From 120aa49580e6b6979d3ffe434a5b14118db5ec8a Mon Sep 17 00:00:00 2001 From: CJ Pais Date: Sat, 3 Oct 2026 21:58:58 +0800 Subject: [PATCH 1/8] ggml: add backend registration filter patch --- ggml/UPSTREAM | 1 + ggml/include/ggml-backend.h | 11 ++ ggml/src/ggml-backend-reg.cpp | 80 ++++++++-- patches/ggml/0002-backend-reg-filter.patch | 177 +++++++++++++++++++++ 4 files changed, 252 insertions(+), 17 deletions(-) create mode 100644 patches/ggml/0002-backend-reg-filter.patch diff --git a/ggml/UPSTREAM b/ggml/UPSTREAM index 78685aafe..c7613871e 100644 --- a/ggml/UPSTREAM +++ b/ggml/UPSTREAM @@ -2,6 +2,7 @@ repo: git@github.com:ggml-org/ggml.git sha: 353b63b439f27ab2cc19dac97ab1681ba6d2d084 patches: patches/ggml/0001-fix-threadpool-oversubscription.patch + patches/ggml/0002-backend-reg-filter.patch This directory is generated from the upstream ggml tree at the SHA above, minus .github/, with the listed downstream patches applied in order. Do not edit it by diff --git a/ggml/include/ggml-backend.h b/ggml/include/ggml-backend.h index cc3f8cd36..dd98d879c 100644 --- a/ggml/include/ggml-backend.h +++ b/ggml/include/ggml-backend.h @@ -259,6 +259,17 @@ extern "C" { GGML_API void ggml_backend_load_all(void); GGML_API void ggml_backend_load_all_from_path(const char * dir_path); + // Registration filter (transcribe.cpp downstream patch). When set, the + // registry consults it before registering each compiled-in backend and + // before opening any dynamic backend module, keyed by the lower-case + // module name ("cpu", "blas", "metal", "vulkan", "cuda", "hip", ...; + // "external" for GGML_BACKEND_PATH). A rejected backend's code never runs: + // no reg function call, no dlopen. Explicit ggml_backend_load(path) calls + // are not filtered. Install before the first registry access so the + // compiled-in backends see it. NULL (the default) allows everything. + typedef bool (*ggml_backend_reg_filter_t)(const char * name); + GGML_API void ggml_backend_set_reg_filter(ggml_backend_reg_filter_t filter); + // // Backend scheduler // diff --git a/ggml/src/ggml-backend-reg.cpp b/ggml/src/ggml-backend-reg.cpp index 1c18b82cd..068bfe897 100644 --- a/ggml/src/ggml-backend-reg.cpp +++ b/ggml/src/ggml-backend-reg.cpp @@ -3,6 +3,7 @@ #include "ggml-backend-dl.h" #include "ggml-impl.h" #include +#include #include #include #include @@ -107,6 +108,17 @@ static std::string path_str(const fs::path & path) { } } +static std::atomic g_reg_filter{nullptr}; + +static bool reg_allowed(const char * name) { + ggml_backend_reg_filter_t filter = g_reg_filter.load(); + return filter == nullptr || filter(name); +} + +void ggml_backend_set_reg_filter(ggml_backend_reg_filter_t filter) { + g_reg_filter.store(filter); +} + struct ggml_backend_reg_entry { ggml_backend_reg_t reg; dl_handle_ptr handle; @@ -118,58 +130,88 @@ struct ggml_backend_registry { ggml_backend_registry() { #ifdef GGML_USE_CUDA - register_backend(ggml_backend_cuda_reg()); + if (reg_allowed("cuda")) { + register_backend(ggml_backend_cuda_reg()); + } #endif #ifdef GGML_USE_METAL - register_backend(ggml_backend_metal_reg()); + if (reg_allowed("metal")) { + register_backend(ggml_backend_metal_reg()); + } #endif #ifdef GGML_USE_SYCL - register_backend(ggml_backend_sycl_reg()); + if (reg_allowed("sycl")) { + register_backend(ggml_backend_sycl_reg()); + } #endif #ifdef GGML_USE_VULKAN // Add runtime disable check - if (getenv("GGML_DISABLE_VULKAN") == nullptr) { + if (getenv("GGML_DISABLE_VULKAN") == nullptr && reg_allowed("vulkan")) { register_backend(ggml_backend_vk_reg()); } else { GGML_LOG_DEBUG("Vulkan backend disabled by GGML_DISABLE_VULKAN environment variable\n"); } #endif #ifdef GGML_USE_WEBGPU - register_backend(ggml_backend_webgpu_reg()); + if (reg_allowed("webgpu")) { + register_backend(ggml_backend_webgpu_reg()); + } #endif #ifdef GGML_USE_ZDNN - register_backend(ggml_backend_zdnn_reg()); + if (reg_allowed("zdnn")) { + register_backend(ggml_backend_zdnn_reg()); + } #endif #ifdef GGML_USE_VIRTGPU_FRONTEND - register_backend(ggml_backend_virtgpu_reg()); + if (reg_allowed("virtgpu")) { + register_backend(ggml_backend_virtgpu_reg()); + } #endif #ifdef GGML_USE_OPENCL - register_backend(ggml_backend_opencl_reg()); + if (reg_allowed("opencl")) { + register_backend(ggml_backend_opencl_reg()); + } #endif #ifdef GGML_USE_ZENDNN - register_backend(ggml_backend_zendnn_reg()); + if (reg_allowed("zendnn")) { + register_backend(ggml_backend_zendnn_reg()); + } #endif #ifdef GGML_USE_HEXAGON - register_backend(ggml_backend_hexagon_reg()); + if (reg_allowed("hexagon")) { + register_backend(ggml_backend_hexagon_reg()); + } #endif #ifdef GGML_USE_CANN - register_backend(ggml_backend_cann_reg()); + if (reg_allowed("cann")) { + register_backend(ggml_backend_cann_reg()); + } #endif #ifdef GGML_USE_BLAS - register_backend(ggml_backend_blas_reg()); + if (reg_allowed("blas")) { + register_backend(ggml_backend_blas_reg()); + } #endif #ifdef GGML_USE_RPC - register_backend(ggml_backend_rpc_reg()); + if (reg_allowed("rpc")) { + register_backend(ggml_backend_rpc_reg()); + } #endif #ifdef GGML_USE_OPENVINO - register_backend(ggml_backend_openvino_reg()); + if (reg_allowed("openvino")) { + register_backend(ggml_backend_openvino_reg()); + } #endif #ifdef GGML_USE_ET - register_backend(ggml_backend_et_reg()); + if (reg_allowed("et")) { + register_backend(ggml_backend_et_reg()); + } #endif #ifdef GGML_USE_CPU - register_backend(ggml_backend_cpu_reg()); + if (reg_allowed("cpu")) { + register_backend(ggml_backend_cpu_reg()); + } #endif } @@ -478,6 +520,10 @@ static fs::path backend_filename_extension() { } static ggml_backend_reg_t ggml_backend_load_best(const char * name, bool silent, const char * user_search_path) { + if (!reg_allowed(name)) { + return nullptr; + } + // enumerate all the files that match [lib]ggml-name-*.[so|dll] in the search paths const fs::path name_path = fs::u8path(name); const fs::path file_prefix = backend_filename_prefix().native() + name_path.native() + fs::u8path("-").native(); @@ -599,7 +645,7 @@ void ggml_backend_load_all_from_path(const char * dir_path) { ggml_backend_load_best("cpu", silent, dir_path); // check the environment variable GGML_BACKEND_PATH to load an out-of-tree backend const char * backend_path = std::getenv("GGML_BACKEND_PATH"); - if (backend_path) { + if (backend_path && reg_allowed("external")) { ggml_backend_load(backend_path); } } diff --git a/patches/ggml/0002-backend-reg-filter.patch b/patches/ggml/0002-backend-reg-filter.patch new file mode 100644 index 000000000..5b7a21421 --- /dev/null +++ b/patches/ggml/0002-backend-reg-filter.patch @@ -0,0 +1,177 @@ +diff --git a/include/ggml-backend.h b/include/ggml-backend.h +index cc3f8cd3..dd98d879 100644 +--- a/include/ggml-backend.h ++++ b/include/ggml-backend.h +@@ -259,6 +259,17 @@ extern "C" { + GGML_API void ggml_backend_load_all(void); + GGML_API void ggml_backend_load_all_from_path(const char * dir_path); + ++ // Registration filter (transcribe.cpp downstream patch). When set, the ++ // registry consults it before registering each compiled-in backend and ++ // before opening any dynamic backend module, keyed by the lower-case ++ // module name ("cpu", "blas", "metal", "vulkan", "cuda", "hip", ...; ++ // "external" for GGML_BACKEND_PATH). A rejected backend's code never runs: ++ // no reg function call, no dlopen. Explicit ggml_backend_load(path) calls ++ // are not filtered. Install before the first registry access so the ++ // compiled-in backends see it. NULL (the default) allows everything. ++ typedef bool (*ggml_backend_reg_filter_t)(const char * name); ++ GGML_API void ggml_backend_set_reg_filter(ggml_backend_reg_filter_t filter); ++ + // + // Backend scheduler + // +diff --git a/src/ggml-backend-reg.cpp b/src/ggml-backend-reg.cpp +index 1c18b82c..068bfe89 100644 +--- a/src/ggml-backend-reg.cpp ++++ b/src/ggml-backend-reg.cpp +@@ -3,6 +3,7 @@ + #include "ggml-backend-dl.h" + #include "ggml-impl.h" + #include ++#include + #include + #include + #include +@@ -107,6 +108,17 @@ static std::string path_str(const fs::path & path) { + } + } + ++static std::atomic g_reg_filter{nullptr}; ++ ++static bool reg_allowed(const char * name) { ++ ggml_backend_reg_filter_t filter = g_reg_filter.load(); ++ return filter == nullptr || filter(name); ++} ++ ++void ggml_backend_set_reg_filter(ggml_backend_reg_filter_t filter) { ++ g_reg_filter.store(filter); ++} ++ + struct ggml_backend_reg_entry { + ggml_backend_reg_t reg; + dl_handle_ptr handle; +@@ -118,58 +130,88 @@ struct ggml_backend_registry { + + ggml_backend_registry() { + #ifdef GGML_USE_CUDA +- register_backend(ggml_backend_cuda_reg()); ++ if (reg_allowed("cuda")) { ++ register_backend(ggml_backend_cuda_reg()); ++ } + #endif + #ifdef GGML_USE_METAL +- register_backend(ggml_backend_metal_reg()); ++ if (reg_allowed("metal")) { ++ register_backend(ggml_backend_metal_reg()); ++ } + #endif + #ifdef GGML_USE_SYCL +- register_backend(ggml_backend_sycl_reg()); ++ if (reg_allowed("sycl")) { ++ register_backend(ggml_backend_sycl_reg()); ++ } + #endif + #ifdef GGML_USE_VULKAN + // Add runtime disable check +- if (getenv("GGML_DISABLE_VULKAN") == nullptr) { ++ if (getenv("GGML_DISABLE_VULKAN") == nullptr && reg_allowed("vulkan")) { + register_backend(ggml_backend_vk_reg()); + } else { + GGML_LOG_DEBUG("Vulkan backend disabled by GGML_DISABLE_VULKAN environment variable\n"); + } + #endif + #ifdef GGML_USE_WEBGPU +- register_backend(ggml_backend_webgpu_reg()); ++ if (reg_allowed("webgpu")) { ++ register_backend(ggml_backend_webgpu_reg()); ++ } + #endif + #ifdef GGML_USE_ZDNN +- register_backend(ggml_backend_zdnn_reg()); ++ if (reg_allowed("zdnn")) { ++ register_backend(ggml_backend_zdnn_reg()); ++ } + #endif + #ifdef GGML_USE_VIRTGPU_FRONTEND +- register_backend(ggml_backend_virtgpu_reg()); ++ if (reg_allowed("virtgpu")) { ++ register_backend(ggml_backend_virtgpu_reg()); ++ } + #endif + + #ifdef GGML_USE_OPENCL +- register_backend(ggml_backend_opencl_reg()); ++ if (reg_allowed("opencl")) { ++ register_backend(ggml_backend_opencl_reg()); ++ } + #endif + #ifdef GGML_USE_ZENDNN +- register_backend(ggml_backend_zendnn_reg()); ++ if (reg_allowed("zendnn")) { ++ register_backend(ggml_backend_zendnn_reg()); ++ } + #endif + #ifdef GGML_USE_HEXAGON +- register_backend(ggml_backend_hexagon_reg()); ++ if (reg_allowed("hexagon")) { ++ register_backend(ggml_backend_hexagon_reg()); ++ } + #endif + #ifdef GGML_USE_CANN +- register_backend(ggml_backend_cann_reg()); ++ if (reg_allowed("cann")) { ++ register_backend(ggml_backend_cann_reg()); ++ } + #endif + #ifdef GGML_USE_BLAS +- register_backend(ggml_backend_blas_reg()); ++ if (reg_allowed("blas")) { ++ register_backend(ggml_backend_blas_reg()); ++ } + #endif + #ifdef GGML_USE_RPC +- register_backend(ggml_backend_rpc_reg()); ++ if (reg_allowed("rpc")) { ++ register_backend(ggml_backend_rpc_reg()); ++ } + #endif + #ifdef GGML_USE_OPENVINO +- register_backend(ggml_backend_openvino_reg()); ++ if (reg_allowed("openvino")) { ++ register_backend(ggml_backend_openvino_reg()); ++ } + #endif + #ifdef GGML_USE_ET +- register_backend(ggml_backend_et_reg()); ++ if (reg_allowed("et")) { ++ register_backend(ggml_backend_et_reg()); ++ } + #endif + #ifdef GGML_USE_CPU +- register_backend(ggml_backend_cpu_reg()); ++ if (reg_allowed("cpu")) { ++ register_backend(ggml_backend_cpu_reg()); ++ } + #endif + } + +@@ -478,6 +520,10 @@ static fs::path backend_filename_extension() { + } + + static ggml_backend_reg_t ggml_backend_load_best(const char * name, bool silent, const char * user_search_path) { ++ if (!reg_allowed(name)) { ++ return nullptr; ++ } ++ + // enumerate all the files that match [lib]ggml-name-*.[so|dll] in the search paths + const fs::path name_path = fs::u8path(name); + const fs::path file_prefix = backend_filename_prefix().native() + name_path.native() + fs::u8path("-").native(); +@@ -599,7 +645,7 @@ void ggml_backend_load_all_from_path(const char * dir_path) { + ggml_backend_load_best("cpu", silent, dir_path); + // check the environment variable GGML_BACKEND_PATH to load an out-of-tree backend + const char * backend_path = std::getenv("GGML_BACKEND_PATH"); +- if (backend_path) { ++ if (backend_path && reg_allowed("external")) { + ggml_backend_load(backend_path); + } + } From 75839c10d4a44609b96aa7523f75ce6c644c19a0 Mon Sep 17 00:00:00 2001 From: CJ Pais Date: Sat, 3 Oct 2026 22:07:43 +0800 Subject: [PATCH 2/8] ggml: report reg-filter vulkan skip accurately --- ggml/src/ggml-backend-reg.cpp | 6 +++--- patches/ggml/0002-backend-reg-filter.patch | 10 ++++++---- 2 files changed, 9 insertions(+), 7 deletions(-) diff --git a/ggml/src/ggml-backend-reg.cpp b/ggml/src/ggml-backend-reg.cpp index 068bfe897..b2a8c193c 100644 --- a/ggml/src/ggml-backend-reg.cpp +++ b/ggml/src/ggml-backend-reg.cpp @@ -146,10 +146,10 @@ struct ggml_backend_registry { #endif #ifdef GGML_USE_VULKAN // Add runtime disable check - if (getenv("GGML_DISABLE_VULKAN") == nullptr && reg_allowed("vulkan")) { - register_backend(ggml_backend_vk_reg()); - } else { + if (getenv("GGML_DISABLE_VULKAN") != nullptr) { GGML_LOG_DEBUG("Vulkan backend disabled by GGML_DISABLE_VULKAN environment variable\n"); + } else if (reg_allowed("vulkan")) { + register_backend(ggml_backend_vk_reg()); } #endif #ifdef GGML_USE_WEBGPU diff --git a/patches/ggml/0002-backend-reg-filter.patch b/patches/ggml/0002-backend-reg-filter.patch index 5b7a21421..0ac12f3c5 100644 --- a/patches/ggml/0002-backend-reg-filter.patch +++ b/patches/ggml/0002-backend-reg-filter.patch @@ -21,7 +21,7 @@ index cc3f8cd3..dd98d879 100644 // Backend scheduler // diff --git a/src/ggml-backend-reg.cpp b/src/ggml-backend-reg.cpp -index 1c18b82c..068bfe89 100644 +index 1c18b82c..b2a8c193 100644 --- a/src/ggml-backend-reg.cpp +++ b/src/ggml-backend-reg.cpp @@ -3,6 +3,7 @@ @@ -74,10 +74,12 @@ index 1c18b82c..068bfe89 100644 #ifdef GGML_USE_VULKAN // Add runtime disable check - if (getenv("GGML_DISABLE_VULKAN") == nullptr) { -+ if (getenv("GGML_DISABLE_VULKAN") == nullptr && reg_allowed("vulkan")) { - register_backend(ggml_backend_vk_reg()); - } else { +- register_backend(ggml_backend_vk_reg()); +- } else { ++ if (getenv("GGML_DISABLE_VULKAN") != nullptr) { GGML_LOG_DEBUG("Vulkan backend disabled by GGML_DISABLE_VULKAN environment variable\n"); ++ } else if (reg_allowed("vulkan")) { ++ register_backend(ggml_backend_vk_reg()); } #endif #ifdef GGML_USE_WEBGPU From fb5e1d3e8e3815c4a8a3f5307e665a2c3a266cda Mon Sep 17 00:00:00 2001 From: CJ Pais Date: Sat, 3 Oct 2026 22:07:43 +0800 Subject: [PATCH 3/8] add allowed-backend mask: transcribe_init_backends_ex + TRANSCRIBE_BACKENDS --- include/transcribe.h | 78 ++++++++++++++++ src/transcribe.cpp | 188 ++++++++++++++++++++++++++++++++++++++ tests/CMakeLists.txt | 23 +++++ tests/backend_mask_unit.c | 175 +++++++++++++++++++++++++++++++++++ 4 files changed, 464 insertions(+) create mode 100644 tests/backend_mask_unit.c diff --git a/include/transcribe.h b/include/transcribe.h index 1a9a17b28..865cfb110 100644 --- a/include/transcribe.h +++ b/include/transcribe.h @@ -825,6 +825,84 @@ TRANSCRIBE_API transcribe_status transcribe_init_backends(const char * artifact_ */ TRANSCRIBE_API transcribe_status transcribe_init_backends_default(void); +/* + * Restricting which backends may initialize. + * + * Registering a GPU backend runs driver code: Vulkan creates an instance + * (loading every installed ICD), Metal opens the system device, CUDA + * initializes the driver. A broken driver can crash or hang the process + * right there, before any model is loaded. A host that isolates inference + * in a worker process can recover from that only if the replacement worker + * never runs the failing backend's code at all — hiding its devices after + * registration is too late. + * + * transcribe_init_backends_ex() takes an allow-mask of backend kinds. A + * backend outside the mask is never registered: its module is never opened + * (dynamic-backend builds) and its registration function is never called + * (static builds). The CPU backend is always allowed, so a mask of 0 or + * TRANSCRIBE_BACKEND_MASK_CPU means "CPU only". + * + * CPU the CPU backend plus host-memory accelerators (BLAS, ZenDNN) + * METAL Apple Metal + * VULKAN Vulkan + * CUDA NVIDIA CUDA + * ROCM AMD ROCm / HIP + * OTHER every backend without a dedicated bit in the running library + * (SYCL, OpenCL, RPC, ..., and an out-of-tree module named by + * GGML_BACKEND_PATH) + * + * The TRANSCRIBE_BACKENDS environment variable can only narrow the mask + * further: a comma-separated list of cpu, metal, vulkan, cuda, rocm, other, + * all (case-insensitive; e.g. TRANSCRIBE_BACKENDS=cpu forces CPU-only + * regardless of what the host passes). Unset or empty means "all". Unknown + * names are logged and ignored. It applies to every way backends get + * registered, including hosts that never call this function. + * + * The mask is FIXED the first time the library registers backends — that + * is, the first transcribe_init_backends*() call or, in static builds, the + * first call that enumerates devices or loads a model. Backends cannot be + * unregistered, so a later call asking for a different effective mask + * returns TRANSCRIBE_ERR_BACKEND without changing anything. Call this once, + * first, per process. + */ +#define TRANSCRIBE_BACKEND_MASK_CPU (1u << 0) +#define TRANSCRIBE_BACKEND_MASK_METAL (1u << 1) +#define TRANSCRIBE_BACKEND_MASK_VULKAN (1u << 2) +#define TRANSCRIBE_BACKEND_MASK_CUDA (1u << 3) +#define TRANSCRIBE_BACKEND_MASK_ROCM (1u << 4) +#define TRANSCRIBE_BACKEND_MASK_OTHER (1u << 31) +#define TRANSCRIBE_BACKEND_MASK_ALL 0xFFFFFFFFu + +struct transcribe_backend_init_params { + uint64_t struct_size; /* sizeof(*this); set by _init() */ + const char * artifact_dir; /* NULL: package-local default, as + transcribe_init_backends_default() */ + uint32_t allowed_backends; /* TRANSCRIBE_BACKEND_MASK_* bits; + _init() sets ..._MASK_ALL */ +}; + +TRANSCRIBE_API void transcribe_backend_init_params_init(struct transcribe_backend_init_params * p); + +/* + * Fix the allowed-backend mask (see above), then load backend modules as + * transcribe_init_backends(artifact_dir) or, with artifact_dir NULL, + * transcribe_init_backends_default() would. NULL params means all defaults. + * + * Returns the statuses of those calls, plus: + * TRANSCRIBE_ERR_BAD_STRUCT_SIZE params fails the struct-size check. + * TRANSCRIBE_ERR_BACKEND the mask was already fixed to a different + * effective value, or no compute device is + * registered afterwards. + */ +TRANSCRIBE_API transcribe_status transcribe_init_backends_ex(const struct transcribe_backend_init_params * params); + +/* + * The effective allowed-backend mask: the host's mask (ALL until + * transcribe_init_backends_ex() sets one) narrowed by TRANSCRIBE_BACKENDS, + * with the CPU bit always set. + */ +TRANSCRIBE_API uint32_t transcribe_allowed_backends(void); + /* * Opaque process-local compute-device handle. Handles are owned by the * runtime, remain valid for the life of the process, and must not be freed. diff --git a/src/transcribe.cpp b/src/transcribe.cpp index 852702d60..9f3a3306e 100644 --- a/src/transcribe.cpp +++ b/src/transcribe.cpp @@ -45,12 +45,14 @@ #include #include +#include #include #include #include #include #include #include +#include #include #include #include @@ -785,6 +787,15 @@ extern "C" void transcribe_device_info_init(struct transcribe_device_info * p) { p->struct_size = sizeof(*p); } +extern "C" void transcribe_backend_init_params_init(struct transcribe_backend_init_params * p) { + if (p == nullptr) { + return; + } + std::memset(p, 0, sizeof(*p)); + p->struct_size = sizeof(*p); + p->allowed_backends = TRANSCRIBE_BACKEND_MASK_ALL; +} + extern "C" void transcribe_word_init(struct transcribe_word * p) { if (p == nullptr) { return; @@ -891,6 +902,8 @@ constexpr size_t k_min_token_size = TRANSCRIBE_FIELD_END(transcribe_to constexpr size_t k_min_speaker_segment_size = TRANSCRIBE_FIELD_END(transcribe_speaker_segment, p); constexpr size_t k_min_timings_size = TRANSCRIBE_FIELD_END(transcribe_timings, decode_ms); constexpr size_t k_min_device_info_size = TRANSCRIBE_FIELD_END(transcribe_device_info, kind); +constexpr size_t k_min_backend_init_params_size = + TRANSCRIBE_FIELD_END(transcribe_backend_init_params, allowed_backends); // k_min_whisper_chunk_trace_size lives in arch/whisper/public.cpp with // the chunk-trace accessor that uses it. @@ -1006,6 +1019,142 @@ static std::string path_for_c_api(const std::filesystem::path & path) { } #endif +// Allowed-backend mask. ggml consults backend_reg_filter (installed at static +// init, before anything can touch ggml's registry) before registering a +// compiled-in backend or opening a backend module, so a backend outside the +// mask never runs any code. The first filter call is the moment backends get +// registered; from then on the mask is fixed (registrations are permanent). +static std::atomic s_host_backend_mask{ TRANSCRIBE_BACKEND_MASK_ALL }; +static std::atomic s_backend_mask_fixed{ false }; +static std::mutex s_backend_mask_mutex; + +static bool ascii_iequals(const char * a, const char * b) { + for (; *a != '\0' && *b != '\0'; ++a, ++b) { + if (std::tolower(static_cast(*a)) != std::tolower(static_cast(*b))) { + return false; + } + } + return *a == *b; +} + +struct BackendMaskName { + const char * name; + uint32_t bit; +}; + +// ggml module names (the [lib]ggml- stem) -> mask bit. Anything not +// listed, including "external" (GGML_BACKEND_PATH), is OTHER. +static uint32_t module_mask_bit(const char * name) { + static const BackendMaskName k_modules[] = { + { "cpu", TRANSCRIBE_BACKEND_MASK_CPU }, + { "blas", TRANSCRIBE_BACKEND_MASK_CPU }, + { "zendnn", TRANSCRIBE_BACKEND_MASK_CPU }, + { "metal", TRANSCRIBE_BACKEND_MASK_METAL }, + { "vulkan", TRANSCRIBE_BACKEND_MASK_VULKAN }, + { "cuda", TRANSCRIBE_BACKEND_MASK_CUDA }, + { "hip", TRANSCRIBE_BACKEND_MASK_ROCM }, + }; + for (const auto & e : k_modules) { + if (ascii_iequals(name, e.name)) { + return e.bit; + } + } + return TRANSCRIBE_BACKEND_MASK_OTHER; +} + +// TRANSCRIBE_BACKENDS, parsed once. Unset or empty is inert (ALL). +static uint32_t env_backend_mask() { + static const uint32_t mask = [] { + static const BackendMaskName k_tokens[] = { + { "cpu", TRANSCRIBE_BACKEND_MASK_CPU }, + { "metal", TRANSCRIBE_BACKEND_MASK_METAL }, + { "vulkan", TRANSCRIBE_BACKEND_MASK_VULKAN }, + { "cuda", TRANSCRIBE_BACKEND_MASK_CUDA }, + { "rocm", TRANSCRIBE_BACKEND_MASK_ROCM }, + { "other", TRANSCRIBE_BACKEND_MASK_OTHER }, + { "all", TRANSCRIBE_BACKEND_MASK_ALL }, + }; + const char * env = std::getenv("TRANSCRIBE_BACKENDS"); + if (env == nullptr || env[0] == '\0') { + return TRANSCRIBE_BACKEND_MASK_ALL; + } + uint32_t m = 0; + std::string tok; + for (const char * p = env;; ++p) { + if (*p != '\0' && *p != ',' && *p != ' ' && *p != '\t') { + tok.push_back(*p); + continue; + } + if (!tok.empty()) { + uint32_t bit = 0; + for (const auto & e : k_tokens) { + if (ascii_iequals(tok.c_str(), e.name)) { + bit = e.bit; + } + } + if (bit == 0) { + transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_WARN, + "TRANSCRIBE_BACKENDS: ignoring unknown backend '%s' " + "(expected cpu, metal, vulkan, cuda, rocm, other, all)", + tok.c_str()); + } + m |= bit; + tok.clear(); + } + if (*p == '\0') { + break; + } + } + return m; + }(); + return mask; +} + +static uint32_t effective_backend_mask(uint32_t host_mask) { + return (host_mask & env_backend_mask()) | TRANSCRIBE_BACKEND_MASK_CPU; +} + +// Called by ggml's registry, possibly from inside its function-local static +// constructor: must not touch the registry, must not throw. +static bool backend_reg_filter(const char * name) noexcept { + try { + s_backend_mask_fixed.store(true); + const uint32_t bit = module_mask_bit(name != nullptr ? name : ""); + const bool allowed = (effective_backend_mask(s_host_backend_mask.load()) & bit) != 0; + if (!allowed) { + transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_DEBUG, + "backend '%s' not registered: excluded by allowed-backend mask", + name != nullptr ? name : "(null)"); + } + return allowed; + } catch (...) { + return false; + } +} + +static const bool s_backend_reg_filter_installed = [] { + ggml_backend_set_reg_filter(&backend_reg_filter); + return true; +}(); + +static transcribe_status set_host_backend_mask(uint32_t mask) { + std::lock_guard lock(s_backend_mask_mutex); + if (s_backend_mask_fixed.load()) { + const uint32_t current = effective_backend_mask(s_host_backend_mask.load()); + const uint32_t wanted = effective_backend_mask(mask); + if (current != wanted) { + transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, + "transcribe_init_backends_ex: allowed-backend mask is already fixed at 0x%08x " + "(backends were registered earlier in this process); cannot change it to 0x%08x", + current, wanted); + return TRANSCRIBE_ERR_BACKEND; + } + return TRANSCRIBE_OK; + } + s_host_backend_mask.store(mask); + return TRANSCRIBE_OK; +} + static transcribe_status transcribe_init_backends_impl(const char * artifact_dir) { ensure_ggml_time_init(); if (artifact_dir == nullptr || artifact_dir[0] == '\0') { @@ -1088,6 +1237,36 @@ static transcribe_status transcribe_init_backends_default_impl(void) { #endif } +static transcribe_status transcribe_init_backends_ex_impl(const struct transcribe_backend_init_params * params) { + ensure_ggml_time_init(); + struct transcribe_backend_init_params p; + transcribe_backend_init_params_init(&p); + if (params != nullptr) { + if (const auto st = check_input_struct_size(params->struct_size, k_min_backend_init_params_size); + st != TRANSCRIBE_OK) { + return st; + } + std::memcpy(&p, params, std::min(params->struct_size, sizeof(p))); + } + if (const auto st = set_host_backend_mask(p.allowed_backends); st != TRANSCRIBE_OK) { + return st; + } + const auto st = p.artifact_dir != nullptr ? transcribe_init_backends_impl(p.artifact_dir) : + transcribe_init_backends_default_impl(); + if (st != TRANSCRIBE_OK) { + return st; + } + // Static builds register compiled-in backends lazily on first registry + // access; force it here so the mask is fixed by this call, as documented. + if (ggml_backend_dev_count() == 0) { + transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, + "transcribe_init_backends_ex: no compute devices registered (allowed-backend mask 0x%08x)", + effective_backend_mask(s_host_backend_mask.load())); + return TRANSCRIBE_ERR_BACKEND; + } + return TRANSCRIBE_OK; +} + namespace { // Map ggml's device-type enum onto the public transcribe_device_type. GPU @@ -3320,6 +3499,15 @@ extern "C" transcribe_status transcribe_init_backends_default(void) { [&] { return transcribe_init_backends_default_impl(); }); } +extern "C" transcribe_status transcribe_init_backends_ex(const struct transcribe_backend_init_params * params) { + return api_guard_status("transcribe_init_backends_ex", [&] { return transcribe_init_backends_ex_impl(params); }); +} + +extern "C" uint32_t transcribe_allowed_backends(void) { + return api_guard_value("transcribe_allowed_backends", static_cast(TRANSCRIBE_BACKEND_MASK_CPU), + [&] { return effective_backend_mask(s_host_backend_mask.load()); }); +} + extern "C" int transcribe_device_count(void) { return api_guard_value("transcribe_device_count", 0, [&] { return transcribe_device_count_impl(); }); } diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 893375542..3ec7871f9 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -54,6 +54,29 @@ target_link_libraries(transcribe_api_smoke PRIVATE transcribe) add_test(NAME transcribe_api_smoke COMMAND transcribe_api_smoke) +# Allowed-backend mask (transcribe_init_backends_ex). The mask is fixed per +# process, so each scenario is its own test invocation. +add_executable(transcribe_backend_mask_unit + backend_mask_unit.c +) + +set_target_properties(transcribe_backend_mask_unit PROPERTIES + C_STANDARD 11 + C_STANDARD_REQUIRED ON + LINKER_LANGUAGE C +) + +target_link_libraries(transcribe_backend_mask_unit PRIVATE transcribe) + +foreach(mode cpu default env-cpu late) + add_test(NAME transcribe_backend_mask_unit_${mode} + COMMAND transcribe_backend_mask_unit ${mode}) + set_tests_properties(transcribe_backend_mask_unit_${mode} PROPERTIES + ENVIRONMENT "TRANSCRIBE_TEST_BACKEND_DIR=$") +endforeach() +set_tests_properties(transcribe_backend_mask_unit_env-cpu PROPERTIES + ENVIRONMENT "TRANSCRIBE_TEST_BACKEND_DIR=$;TRANSCRIBE_BACKENDS=cpu") + add_test( NAME transcribe_extension_umbrella_check COMMAND ${CMAKE_COMMAND} diff --git a/tests/backend_mask_unit.c b/tests/backend_mask_unit.c new file mode 100644 index 000000000..8e07d7b35 --- /dev/null +++ b/tests/backend_mask_unit.c @@ -0,0 +1,175 @@ +/* backend_mask_unit.c - allowed-backend mask (transcribe_init_backends_ex). + * + * The mask is fixed for the life of the process, so each scenario runs in + * its own process: ctest registers this binary once per mode (argv[1]). + * + * cpu mask=CPU: only CPU-kind devices register, GPU requests are + * unavailable, a different mask is refused, the same one is + * accepted, and (Linux) no Vulkan ICD was ever loaded. + * default default params: everything allowed; bad struct_size rejected. + * env-cpu TRANSCRIBE_BACKENDS=cpu narrows a host mask of ALL to CPU. + * late a device query before _ex. Static builds register backends on + * that query, so a narrower mask is refused; dynamic builds have + * nothing registered yet, so it is accepted. + * + * TRANSCRIBE_TEST_BACKEND_DIR names the backend module directory (set by + * ctest to the build's bin dir; static builds scan it as a no-op). + * + * Plain C11 on purpose, like api_smoke.c: uses only the public header. + */ + +#include "transcribe.h" + +#include +#include +#include + +static int g_failures = 0; + +#define CHECK(cond) \ + do { \ + if (!(cond)) { \ + fprintf(stderr, "FAIL %s:%d: %s\n", __FILE__, __LINE__, #cond); \ + ++g_failures; \ + } \ + } while (0) + +static int only_cpu_kind_devices(void) { + const int n = transcribe_device_count(); + for (int i = 0; i < n; ++i) { + struct transcribe_device_info info; + transcribe_device_info_init(&info); + if (transcribe_device_get_info(transcribe_device_get(i), &info) != TRANSCRIBE_OK) { + return 0; + } + if (strcmp(info.kind, "cpu") != 0 && strcmp(info.kind, "accel") != 0) { + fprintf(stderr, "unexpected device %s (kind %s)\n", info.name, info.kind); + return 0; + } + } + return n > 0; +} + +/* Linux Vulkan ICDs are libvulkan_.so (radeon, lvp, intel, ...) and + * are only loaded by vkCreateInstance / instance enumeration. Their absence + * from the address space proves the Vulkan backend never initialized. */ +static int vulkan_icd_loaded(void) { +#if defined(__linux__) + FILE * f = fopen("/proc/self/maps", "r"); + if (f == NULL) { + return 0; + } + char line[4096]; + int found = 0; + while (fgets(line, sizeof(line), f) != NULL) { + if (strstr(line, "libvulkan_") != NULL) { + found = 1; + break; + } + } + fclose(f); + return found; +#else + return 0; +#endif +} + +static transcribe_status init_with_mask(uint32_t mask) { + struct transcribe_backend_init_params p; + transcribe_backend_init_params_init(&p); + p.artifact_dir = getenv("TRANSCRIBE_TEST_BACKEND_DIR"); + p.allowed_backends = mask; + return transcribe_init_backends_ex(&p); +} + +static void run_cpu(void) { + CHECK(init_with_mask(TRANSCRIBE_BACKEND_MASK_CPU) == TRANSCRIBE_OK); + CHECK(transcribe_allowed_backends() == TRANSCRIBE_BACKEND_MASK_CPU); + CHECK(only_cpu_kind_devices()); + CHECK(transcribe_backend_available(TRANSCRIBE_BACKEND_CPU)); + CHECK(!transcribe_backend_available(TRANSCRIBE_BACKEND_VULKAN)); + CHECK(!transcribe_backend_available(TRANSCRIBE_BACKEND_METAL)); + CHECK(!transcribe_backend_available(TRANSCRIBE_BACKEND_CUDA)); + CHECK(!vulkan_icd_loaded()); + + /* Fixed: a different effective mask is refused, the same one (including + * 0, which is CPU-implied) is accepted, and the legacy entry points keep + * the fixed mask. */ + CHECK(init_with_mask(TRANSCRIBE_BACKEND_MASK_ALL) == TRANSCRIBE_ERR_BACKEND); + CHECK(init_with_mask(TRANSCRIBE_BACKEND_MASK_CPU) == TRANSCRIBE_OK); + CHECK(init_with_mask(0) == TRANSCRIBE_OK); + CHECK(transcribe_init_backends(getenv("TRANSCRIBE_TEST_BACKEND_DIR")) == TRANSCRIBE_OK); + CHECK(transcribe_allowed_backends() == TRANSCRIBE_BACKEND_MASK_CPU); + CHECK(only_cpu_kind_devices()); + CHECK(!vulkan_icd_loaded()); +} + +static void run_default(void) { + struct transcribe_backend_init_params bad; + transcribe_backend_init_params_init(&bad); + bad.struct_size = 8; + CHECK(transcribe_init_backends_ex(&bad) == TRANSCRIBE_ERR_BAD_STRUCT_SIZE); + + struct transcribe_backend_init_params p; + transcribe_backend_init_params_init(&p); + p.artifact_dir = getenv("TRANSCRIBE_TEST_BACKEND_DIR"); + CHECK(transcribe_init_backends_ex(&p) == TRANSCRIBE_OK); + CHECK(transcribe_allowed_backends() == TRANSCRIBE_BACKEND_MASK_ALL); + CHECK(transcribe_device_count() > 0); +#if defined(__linux__) + /* Probe self-check: the ICD scan must see a Vulkan backend that did + * initialize, or its absence in the other modes proves nothing. */ + if (transcribe_backend_available(TRANSCRIBE_BACKEND_VULKAN)) { + CHECK(vulkan_icd_loaded()); + } +#endif +} + +static void run_env_cpu(void) { + /* ctest sets TRANSCRIBE_BACKENDS=cpu for this mode. */ + CHECK(init_with_mask(TRANSCRIBE_BACKEND_MASK_ALL) == TRANSCRIBE_OK); + CHECK(transcribe_allowed_backends() == TRANSCRIBE_BACKEND_MASK_CPU); + CHECK(only_cpu_kind_devices()); + CHECK(!vulkan_icd_loaded()); +} + +static void run_late(void) { + const int pre = transcribe_device_count(); + const transcribe_status st = init_with_mask(TRANSCRIBE_BACKEND_MASK_CPU); + if (pre > 0) { + /* Static build: the query already registered everything under ALL, + * so narrowing now is refused. */ + CHECK(st == TRANSCRIBE_ERR_BACKEND); + CHECK(transcribe_allowed_backends() == TRANSCRIBE_BACKEND_MASK_ALL); + } else { + CHECK(st == TRANSCRIBE_OK); + CHECK(transcribe_allowed_backends() == TRANSCRIBE_BACKEND_MASK_CPU); + CHECK(only_cpu_kind_devices()); + } +} + +int main(int argc, char ** argv) { + if (argc != 2) { + fprintf(stderr, "usage: %s cpu|default|env-cpu|late\n", argv[0]); + return EXIT_FAILURE; + } + const char * mode = argv[1]; + if (strcmp(mode, "cpu") == 0) { + run_cpu(); + } else if (strcmp(mode, "default") == 0) { + run_default(); + } else if (strcmp(mode, "env-cpu") == 0) { + run_env_cpu(); + } else if (strcmp(mode, "late") == 0) { + run_late(); + } else { + fprintf(stderr, "unknown mode %s\n", mode); + return EXIT_FAILURE; + } + if (g_failures > 0) { + fprintf(stderr, "backend_mask_unit %s: %d failure(s)\n", mode, g_failures); + return EXIT_FAILURE; + } + printf("backend_mask_unit %s: OK\n", mode); + return EXIT_SUCCESS; +} From a8f1767faacf4b39807e04a6de6309391a9d86f3 Mon Sep 17 00:00:00 2001 From: CJ Pais Date: Sat, 3 Oct 2026 22:08:24 +0800 Subject: [PATCH 4/8] regenerate FFI bindings and abihash --- .../python/src/transcribe_cpp/_generated.py | 13 +++++- bindings/rust/sys/src/transcribe_sys.rs | 42 ++++++++++++++++++- .../swift/Sources/TranscribeCpp/ABIHash.swift | 2 +- bindings/typescript/src/_generated.ts | 8 +++- include/transcribe.abihash | 2 +- 5 files changed, 61 insertions(+), 6 deletions(-) diff --git a/bindings/python/src/transcribe_cpp/_generated.py b/bindings/python/src/transcribe_cpp/_generated.py index d1122d2ae..29b3b2435 100644 --- a/bindings/python/src/transcribe_cpp/_generated.py +++ b/bindings/python/src/transcribe_cpp/_generated.py @@ -13,7 +13,7 @@ # Stable digest of the ABI surface below (structs, enums, macros, layout, # prototypes). A native provider package echoes this back so the API # package can reject an ABI-mismatched provider before dlopen. -PUBLIC_HEADER_HASH = "59b9a92b47074666" +PUBLIC_HEADER_HASH = "8c53a55291ec1597" # === enum constants === TRANSCRIBE_OK = 0 @@ -116,6 +116,7 @@ TRANSCRIBE_WHISPER_PROMPT_ALL_SEGMENTS = 1 # === macro constants (integer object-like macros) === +TRANSCRIBE_BACKEND_MASK_ALL = 4294967295 TRANSCRIBE_EXT_KIND_MOONSHINE_STREAMING_STREAM = 1414746957 TRANSCRIBE_EXT_KIND_PARAKEET_BUFFERED_STREAM = 1396853584 TRANSCRIBE_EXT_KIND_PARAKEET_STREAM = 1414744912 @@ -126,6 +127,8 @@ # === structs === class transcribe_ext(_c.Structure): pass +class transcribe_backend_init_params(_c.Structure): + pass class transcribe_device_info(_c.Structure): pass class transcribe_model_load_params(_c.Structure): @@ -170,6 +173,7 @@ class transcribe_whisper_chunk_trace(_c.Structure): pass transcribe_ext._fields_ = [("size", _c.c_uint64), ("kind", _c.c_uint32)] +transcribe_backend_init_params._fields_ = [("struct_size", _c.c_uint64), ("artifact_dir", _c.c_char_p), ("allowed_backends", _c.c_uint32)] transcribe_device_info._fields_ = [("struct_size", _c.c_uint64), ("name", _c.c_char_p), ("description", _c.c_char_p), ("kind", _c.c_char_p), ("device_id", _c.c_char_p), ("memory_total", _c.c_uint64), ("memory_free", _c.c_uint64), ("device_type", _c.c_int)] transcribe_model_load_params._fields_ = [("struct_size", _c.c_uint64), ("backend", _c.c_int), ("device", _c.c_void_p)] transcribe_session_params._fields_ = [("struct_size", _c.c_uint64), ("n_threads", _c.c_int), ("kv_type", _c.c_int), ("n_ctx", _c.c_int32)] @@ -215,6 +219,7 @@ class transcribe_whisper_chunk_trace(_c.Structure): # C-compiler layout captured at generation (for offset self-check). STRUCT_LAYOUT = { 'transcribe_ext': {'size': 16, 'align': 8, 'offsets': {'size': 0, 'kind': 8}}, + 'transcribe_backend_init_params': {'size': 24, 'align': 8, 'offsets': {'struct_size': 0, 'artifact_dir': 8, 'allowed_backends': 16}}, 'transcribe_device_info': {'size': 64, 'align': 8, 'offsets': {'struct_size': 0, 'name': 8, 'description': 16, 'kind': 24, 'device_id': 32, 'memory_total': 40, 'memory_free': 48, 'device_type': 56}}, 'transcribe_model_load_params': {'size': 24, 'align': 8, 'offsets': {'struct_size': 0, 'backend': 8, 'device': 16}}, 'transcribe_session_params': {'size': 24, 'align': 8, 'offsets': {'struct_size': 0, 'n_threads': 8, 'kv_type': 12, 'n_ctx': 16}}, @@ -245,8 +250,12 @@ def configure(lib): lib.transcribe_abi_struct_align.argtypes = [_c.c_int] lib.transcribe_abi_struct_size.restype = _c.c_size_t lib.transcribe_abi_struct_size.argtypes = [_c.c_int] + lib.transcribe_allowed_backends.restype = _c.c_uint32 + lib.transcribe_allowed_backends.argtypes = [] lib.transcribe_backend_available.restype = _c.c_bool lib.transcribe_backend_available.argtypes = [_c.c_int] + lib.transcribe_backend_init_params_init.restype = None + lib.transcribe_backend_init_params_init.argtypes = [_c.POINTER(transcribe_backend_init_params)] lib.transcribe_batch_detected_language.restype = _c.c_char_p lib.transcribe_batch_detected_language.argtypes = [_c.c_void_p, _c.c_int] lib.transcribe_batch_full_text.restype = _c.c_char_p @@ -315,6 +324,8 @@ def configure(lib): lib.transcribe_init_backends.argtypes = [_c.c_char_p] lib.transcribe_init_backends_default.restype = _c.c_int lib.transcribe_init_backends_default.argtypes = [] + lib.transcribe_init_backends_ex.restype = _c.c_int + lib.transcribe_init_backends_ex.argtypes = [_c.POINTER(transcribe_backend_init_params)] lib.transcribe_log_set.restype = None lib.transcribe_log_set.argtypes = [_c.CFUNCTYPE(None, _c.c_int, _c.c_char_p, _c.c_void_p), _c.c_void_p] lib.transcribe_model_accepts_ext_kind.restype = _c.c_bool diff --git a/bindings/rust/sys/src/transcribe_sys.rs b/bindings/rust/sys/src/transcribe_sys.rs index cbccfab04..7eb810849 100644 --- a/bindings/rust/sys/src/transcribe_sys.rs +++ b/bindings/rust/sys/src/transcribe_sys.rs @@ -1,14 +1,21 @@ // @generated by `cargo xtask bindgen` from include/transcribe/extensions.h // DO NOT EDIT BY HAND. Regenerate: `cargo xtask bindgen`. -// Pinned to include/transcribe.abihash = 59b9a92b47074666 +// Pinned to include/transcribe.abihash = 8c53a55291ec1597 /// The public-ABI digest these bindings were generated against /// (sha256/16 over the normalized FFI surface). The load-time version /// gate and the CI drift check both anchor on this value. -pub const PUBLIC_HEADER_HASH: &str = "59b9a92b47074666"; +pub const PUBLIC_HEADER_HASH: &str = "8c53a55291ec1597"; /* automatically generated by rust-bindgen 0.72.1 */ +pub const TRANSCRIBE_BACKEND_MASK_CPU: u32 = 1; +pub const TRANSCRIBE_BACKEND_MASK_METAL: u32 = 2; +pub const TRANSCRIBE_BACKEND_MASK_VULKAN: u32 = 4; +pub const TRANSCRIBE_BACKEND_MASK_CUDA: u32 = 8; +pub const TRANSCRIBE_BACKEND_MASK_ROCM: u32 = 16; +pub const TRANSCRIBE_BACKEND_MASK_OTHER: u32 = 2147483648; +pub const TRANSCRIBE_BACKEND_MASK_ALL: u32 = 4294967295; pub const TRANSCRIBE_EXT_KIND_MOONSHINE_STREAMING_STREAM: u32 = 1414746957; pub const TRANSCRIBE_EXT_KIND_PARAKEET_STREAM: u32 = 1414744912; pub const TRANSCRIBE_EXT_KIND_PARAKEET_BUFFERED_STREAM: u32 = 1396853584; @@ -217,6 +224,37 @@ unsafe extern "C" { } #[repr(C)] #[derive(Debug, Copy, Clone)] +pub struct transcribe_backend_init_params { + pub struct_size: u64, + pub artifact_dir: *const ::std::os::raw::c_char, + pub allowed_backends: u32, +} +#[allow(clippy::unnecessary_operation, clippy::identity_op)] +const _: () = { + ["Size of transcribe_backend_init_params"] + [::std::mem::size_of::() - 24usize]; + ["Alignment of transcribe_backend_init_params"] + [::std::mem::align_of::() - 8usize]; + ["Offset of field: transcribe_backend_init_params::struct_size"] + [::std::mem::offset_of!(transcribe_backend_init_params, struct_size) - 0usize]; + ["Offset of field: transcribe_backend_init_params::artifact_dir"] + [::std::mem::offset_of!(transcribe_backend_init_params, artifact_dir) - 8usize]; + ["Offset of field: transcribe_backend_init_params::allowed_backends"] + [::std::mem::offset_of!(transcribe_backend_init_params, allowed_backends) - 16usize]; +}; +unsafe extern "C" { + pub fn transcribe_backend_init_params_init(p: *mut transcribe_backend_init_params); +} +unsafe extern "C" { + pub fn transcribe_init_backends_ex( + params: *const transcribe_backend_init_params, + ) -> transcribe_status; +} +unsafe extern "C" { + pub fn transcribe_allowed_backends() -> u32; +} +#[repr(C)] +#[derive(Debug, Copy, Clone)] pub struct transcribe_device { _unused: [u8; 0], } diff --git a/bindings/swift/Sources/TranscribeCpp/ABIHash.swift b/bindings/swift/Sources/TranscribeCpp/ABIHash.swift index f55cc4c77..34f41a228 100644 --- a/bindings/swift/Sources/TranscribeCpp/ABIHash.swift +++ b/bindings/swift/Sources/TranscribeCpp/ABIHash.swift @@ -13,7 +13,7 @@ import CTranscribe extension Transcribe { /// sha256/16 of the normalized public FFI surface, pinned to the value in /// include/transcribe.abihash at the time this binding was last reviewed. - public static let pinnedHeaderHash = "59b9a92b47074666" + public static let pinnedHeaderHash = "8c53a55291ec1597" /// The public-ABI digest this binding was reviewed against (16 hex chars). public static func headerHash() -> String { pinnedHeaderHash } diff --git a/bindings/typescript/src/_generated.ts b/bindings/typescript/src/_generated.ts index 7c4db3ce9..52e980540 100644 --- a/bindings/typescript/src/_generated.ts +++ b/bindings/typescript/src/_generated.ts @@ -11,7 +11,7 @@ // Stable digest of the ABI surface (structs, enums, macros, layout, // prototypes), computed by the Python oracle and pinned here so a header // ABI change turns this binding's drift check red for conscious review. -export const PUBLIC_HEADER_HASH = "59b9a92b47074666"; +export const PUBLIC_HEADER_HASH = "8c53a55291ec1597"; // === enum constants === export const TRANSCRIBE_OK = 0; @@ -114,6 +114,7 @@ export const TRANSCRIBE_WHISPER_PROMPT_FIRST_SEGMENT = 0; export const TRANSCRIBE_WHISPER_PROMPT_ALL_SEGMENTS = 1; // === macro constants (integer object-like macros) === +export const TRANSCRIBE_BACKEND_MASK_ALL = 4294967295; export const TRANSCRIBE_EXT_KIND_MOONSHINE_STREAMING_STREAM = 1414746957; export const TRANSCRIBE_EXT_KIND_PARAKEET_BUFFERED_STREAM = 1396853584; export const TRANSCRIBE_EXT_KIND_PARAKEET_STREAM = 1414744912; @@ -124,6 +125,7 @@ export const TRANSCRIBE_EXT_KIND_WHISPER_RUN = 1314015319; export interface StructLayout { size: number; align: number; offsets: Record; } export const STRUCT_LAYOUT: Record = { 'transcribe_ext': { size: 16, align: 8, offsets: {'size': 0, 'kind': 8} }, + 'transcribe_backend_init_params': { size: 24, align: 8, offsets: {'struct_size': 0, 'artifact_dir': 8, 'allowed_backends': 16} }, 'transcribe_device_info': { size: 64, align: 8, offsets: {'struct_size': 0, 'name': 8, 'description': 16, 'kind': 24, 'device_id': 32, 'memory_total': 40, 'memory_free': 48, 'device_type': 56} }, 'transcribe_model_load_params': { size: 24, align: 8, offsets: {'struct_size': 0, 'backend': 8, 'device': 16} }, 'transcribe_session_params': { size: 24, align: 8, offsets: {'struct_size': 0, 'n_threads': 8, 'kv_type': 12, 'n_ctx': 16} }, @@ -169,6 +171,7 @@ export const ABI_STRUCT_IDS: Record = { export function defineTypes(koffi: any): Record { const T: Record = {}; T['transcribe_ext'] = koffi.struct({ size: 'uint64_t', kind: 'uint32_t' }); + T['transcribe_backend_init_params'] = koffi.struct({ struct_size: 'uint64_t', artifact_dir: 'char *', allowed_backends: 'uint32_t' }); T['transcribe_device_info'] = koffi.struct({ struct_size: 'uint64_t', name: 'char *', description: 'char *', kind: 'char *', device_id: 'char *', memory_total: 'uint64_t', memory_free: 'uint64_t', device_type: 'int' }); T['transcribe_model_load_params'] = koffi.struct({ struct_size: 'uint64_t', backend: 'int', device: 'void *' }); T['transcribe_session_params'] = koffi.struct({ struct_size: 'uint64_t', n_threads: 'int', kv_type: 'int', n_ctx: 'int32_t' }); @@ -197,7 +200,9 @@ export interface FnSig { ret: string; args: string[]; } export const FUNCTION_SIGNATURES: Record = { 'transcribe_abi_struct_align': { ret: 'size_t', args: ['transcribe_abi_struct'] }, 'transcribe_abi_struct_size': { ret: 'size_t', args: ['transcribe_abi_struct'] }, + 'transcribe_allowed_backends': { ret: 'uint32_t', args: [] }, 'transcribe_backend_available': { ret: '_Bool', args: ['transcribe_backend_request'] }, + 'transcribe_backend_init_params_init': { ret: 'void', args: ['struct transcribe_backend_init_params *'] }, 'transcribe_batch_detected_language': { ret: 'const char *', args: ['const struct transcribe_session *', 'int'] }, 'transcribe_batch_full_text': { ret: 'const char *', args: ['const struct transcribe_session *', 'int'] }, 'transcribe_batch_get_segment': { ret: 'transcribe_status', args: ['const struct transcribe_session *', 'int', 'int', 'struct transcribe_segment *'] }, @@ -232,6 +237,7 @@ export const FUNCTION_SIGNATURES: Record = { 'transcribe_get_word': { ret: 'transcribe_status', args: ['const struct transcribe_session *', 'int', 'struct transcribe_word *'] }, 'transcribe_init_backends': { ret: 'transcribe_status', args: ['const char *'] }, 'transcribe_init_backends_default': { ret: 'transcribe_status', args: [] }, + 'transcribe_init_backends_ex': { ret: 'transcribe_status', args: ['const struct transcribe_backend_init_params *'] }, 'transcribe_log_set': { ret: 'void', args: ['transcribe_log_callback', 'void *'] }, 'transcribe_model_accepts_ext_kind': { ret: '_Bool', args: ['const struct transcribe_model *', 'transcribe_ext_slot', 'uint32_t'] }, 'transcribe_model_arch_string': { ret: 'const char *', args: ['const struct transcribe_model *'] }, diff --git a/include/transcribe.abihash b/include/transcribe.abihash index ae00915b6..0083d8b24 100644 --- a/include/transcribe.abihash +++ b/include/transcribe.abihash @@ -1 +1 @@ -59b9a92b47074666 +8c53a55291ec1597 From 3b166a4d55319dc52a34e034ac4c660afda897c8 Mon Sep 17 00:00:00 2001 From: CJ Pais Date: Sat, 3 Oct 2026 22:12:04 +0800 Subject: [PATCH 5/8] rust: BackendMask + init_backends_with --- bindings/rust/transcribe-cpp/src/backend.rs | 122 ++++++++++++++++++ bindings/rust/transcribe-cpp/src/lib.rs | 4 +- .../rust/transcribe-cpp/tests/backend_mask.rs | 44 +++++++ 3 files changed, 168 insertions(+), 2 deletions(-) create mode 100644 bindings/rust/transcribe-cpp/tests/backend_mask.rs diff --git a/bindings/rust/transcribe-cpp/src/backend.rs b/bindings/rust/transcribe-cpp/src/backend.rs index f19979961..aa176ac56 100644 --- a/bindings/rust/transcribe-cpp/src/backend.rs +++ b/bindings/rust/transcribe-cpp/src/backend.rs @@ -157,6 +157,128 @@ pub fn init_backends_default() -> Result<()> { check(status, "init_backends_default") } +/// A set of backend kinds allowed to register in this process. +/// +/// Registering a GPU backend runs driver code (Vulkan instance creation loads +/// every installed driver; Metal and CUDA initialize theirs), so a broken +/// driver can crash or hang the process before any model loads. A backend +/// outside the mask is never registered: its module is never opened and its +/// registration function never runs. The CPU backend is always allowed. +/// +/// The `TRANSCRIBE_BACKENDS` environment variable (e.g. `cpu` or +/// `cpu,vulkan`) can only narrow the mask further. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))] +pub struct BackendMask(u32); + +impl BackendMask { + /// CPU plus host-memory accelerators (BLAS). Always implied. + pub const CPU: BackendMask = BackendMask(sys::TRANSCRIBE_BACKEND_MASK_CPU); + /// Apple Metal. + pub const METAL: BackendMask = BackendMask(sys::TRANSCRIBE_BACKEND_MASK_METAL); + /// Vulkan. + pub const VULKAN: BackendMask = BackendMask(sys::TRANSCRIBE_BACKEND_MASK_VULKAN); + /// NVIDIA CUDA. + pub const CUDA: BackendMask = BackendMask(sys::TRANSCRIBE_BACKEND_MASK_CUDA); + /// AMD ROCm / HIP. + pub const ROCM: BackendMask = BackendMask(sys::TRANSCRIBE_BACKEND_MASK_ROCM); + /// Every backend without a dedicated bit (SYCL, OpenCL, RPC, ...). + pub const OTHER: BackendMask = BackendMask(sys::TRANSCRIBE_BACKEND_MASK_OTHER); + /// Everything. The default. + pub const ALL: BackendMask = BackendMask(sys::TRANSCRIBE_BACKEND_MASK_ALL); + + /// The smallest mask that can satisfy a model-load `backend` request: + /// [`Backend::Auto`] needs everything, [`Backend::Cpu`] / + /// [`Backend::CpuAccel`] only the CPU, and an explicit GPU backend only + /// itself (plus the implied CPU). A worker process that serves exactly + /// one request kind passes this to [`init_backends_with`]. + pub const fn for_backend(backend: Backend) -> BackendMask { + match backend { + Backend::Auto => BackendMask::ALL, + Backend::Cpu | Backend::CpuAccel => BackendMask::CPU, + Backend::Metal => BackendMask::METAL.union(BackendMask::CPU), + Backend::Vulkan => BackendMask::VULKAN.union(BackendMask::CPU), + Backend::Cuda => BackendMask::CUDA.union(BackendMask::CPU), + Backend::Rocm => BackendMask::ROCM.union(BackendMask::CPU), + } + } + + /// The raw `TRANSCRIBE_BACKEND_MASK_*` bits. + pub const fn bits(self) -> u32 { + self.0 + } + + /// A mask from raw `TRANSCRIBE_BACKEND_MASK_*` bits. + pub const fn from_bits(bits: u32) -> BackendMask { + BackendMask(bits) + } + + /// Both masks' backends. + pub const fn union(self, other: BackendMask) -> BackendMask { + BackendMask(self.0 | other.0) + } + + /// Whether every backend in `other` is in `self`. + pub const fn contains(self, other: BackendMask) -> bool { + self.0 & other.0 == other.0 + } +} + +impl Default for BackendMask { + fn default() -> Self { + BackendMask::ALL + } +} + +impl std::ops::BitOr for BackendMask { + type Output = BackendMask; + fn bitor(self, rhs: BackendMask) -> BackendMask { + self.union(rhs) + } +} + +impl std::ops::BitOrAssign for BackendMask { + fn bitor_assign(&mut self, rhs: BackendMask) { + *self = self.union(rhs); + } +} + +/// [`init_backends`] / [`init_backends_default`] with an allowed-backend mask. +/// `dir` is the backend module directory, or `None` for the package-local +/// default. +/// +/// The mask is fixed the first time backends register in this process (this +/// call, or in static builds the first device query or model load). Call it +/// first, once per process: a later call asking for a different effective +/// mask errors with [`crate::Error::Backend`]. Registration is permanent, so a +/// worker that must avoid a backend needs a fresh process. +/// +/// ```no_run +/// use transcribe_cpp::{init_backends_with, Backend, BackendMask}; +/// // A CPU fallback worker: never let the GPU driver load. +/// init_backends_with(None::<&std::path::Path>, BackendMask::for_backend(Backend::Cpu))?; +/// # Ok::<(), transcribe_cpp::Error>(()) +/// ``` +pub fn init_backends_with(dir: Option>, allowed: BackendMask) -> Result<()> { + let c_dir = match dir { + Some(dir) => Some(CString::new(crate::model::path_bytes(dir.as_ref())?)?), + None => None, + }; + let mut params: sys::transcribe_backend_init_params = unsafe { std::mem::zeroed() }; + unsafe { sys::transcribe_backend_init_params_init(&mut params) }; + params.artifact_dir = c_dir.as_ref().map_or(std::ptr::null(), |d| d.as_ptr()); + params.allowed_backends = allowed.bits(); + let status = unsafe { sys::transcribe_init_backends_ex(¶ms) }; + check(status, "init_backends_with") +} + +/// The effective allowed-backend mask: the mask passed to +/// [`init_backends_with`] (everything until then), narrowed by +/// `TRANSCRIBE_BACKENDS`, with the CPU always included. +pub fn allowed_backends() -> BackendMask { + BackendMask(unsafe { sys::transcribe_allowed_backends() }) +} + /// The number of compute devices currently registered. /// /// Do not race this query with [`init_backends`] or [`init_backends_default`]. diff --git a/bindings/rust/transcribe-cpp/src/lib.rs b/bindings/rust/transcribe-cpp/src/lib.rs index d2dc48c76..eb34e0040 100644 --- a/bindings/rust/transcribe-cpp/src/lib.rs +++ b/bindings/rust/transcribe-cpp/src/lib.rs @@ -58,8 +58,8 @@ mod types; mod version; pub use backend::{ - backend_available, device_count, devices, init_backends, init_backends_default, Device, - DeviceType, + allowed_backends, backend_available, device_count, devices, init_backends, + init_backends_default, init_backends_with, BackendMask, Device, DeviceType, }; pub use cancel::CancelToken; pub use error::{Error, Result}; diff --git a/bindings/rust/transcribe-cpp/tests/backend_mask.rs b/bindings/rust/transcribe-cpp/tests/backend_mask.rs new file mode 100644 index 000000000..c4e778040 --- /dev/null +++ b/bindings/rust/transcribe-cpp/tests/backend_mask.rs @@ -0,0 +1,44 @@ +//! Allowed-backend mask. The mask is fixed per process, so every assertion +//! that touches the native registry lives in ONE test function in this file +//! (its own test binary, hence its own process). + +use transcribe_cpp::{ + allowed_backends, backend_available, devices, init_backends_with, Backend, BackendMask, + DeviceType, Error, +}; + +#[test] +fn for_backend_is_minimal() { + assert_eq!(BackendMask::for_backend(Backend::Auto), BackendMask::ALL); + assert_eq!(BackendMask::for_backend(Backend::Cpu), BackendMask::CPU); + assert_eq!(BackendMask::for_backend(Backend::CpuAccel), BackendMask::CPU); + assert_eq!( + BackendMask::for_backend(Backend::Vulkan), + BackendMask::VULKAN | BackendMask::CPU + ); + assert!(BackendMask::ALL.contains(BackendMask::METAL | BackendMask::OTHER)); + assert!(!BackendMask::CPU.contains(BackendMask::VULKAN)); +} + +#[test] +fn cpu_only_worker() { + let cpu = BackendMask::for_backend(Backend::Cpu); + init_backends_with(None::<&std::path::Path>, cpu).expect("cpu-only init"); + assert_eq!(allowed_backends(), BackendMask::CPU); + + let devs = devices(); + assert!(!devs.is_empty()); + assert!(devs + .iter() + .all(|d| matches!(d.device_type, DeviceType::Cpu | DeviceType::Accel))); + assert!(backend_available(Backend::Cpu)); + assert!(!backend_available(Backend::Vulkan)); + assert!(!backend_available(Backend::Metal)); + + // Fixed for the process: the same mask is fine, a different one is not. + init_backends_with(None::<&std::path::Path>, cpu).expect("same mask again"); + assert!(matches!( + init_backends_with(None::<&std::path::Path>, BackendMask::ALL), + Err(Error::Backend { .. }) + )); +} From a4f950df414bc8d667042d36aaabee4c6babac21 Mon Sep 17 00:00:00 2001 From: CJ Pais Date: Sun, 4 Oct 2026 08:23:30 +0800 Subject: [PATCH 6/8] be able to disable backends --- ggml/src/ggml-backend-reg.cpp | 8 ++ ggml/src/ggml-hip/CMakeLists.txt | 2 + ggml/src/ggml-musa/CMakeLists.txt | 2 + patches/ggml/0002-backend-reg-filter.patch | 42 +++++++++- src/transcribe.cpp | 96 +++++++++++++--------- tests/backend_mask_unit.c | 47 +++++++---- 6 files changed, 136 insertions(+), 61 deletions(-) diff --git a/ggml/src/ggml-backend-reg.cpp b/ggml/src/ggml-backend-reg.cpp index b2a8c193c..085ac3034 100644 --- a/ggml/src/ggml-backend-reg.cpp +++ b/ggml/src/ggml-backend-reg.cpp @@ -130,7 +130,15 @@ struct ggml_backend_registry { ggml_backend_registry() { #ifdef GGML_USE_CUDA + // HIP and MUSA builds reuse the CUDA backend; filter them under the + // name their dynamic module would have. +#if defined(GGML_USE_HIP) + if (reg_allowed("hip")) { +#elif defined(GGML_USE_MUSA) + if (reg_allowed("musa")) { +#else if (reg_allowed("cuda")) { +#endif register_backend(ggml_backend_cuda_reg()); } #endif diff --git a/ggml/src/ggml-hip/CMakeLists.txt b/ggml/src/ggml-hip/CMakeLists.txt index a6a6b7271..6140c29dd 100644 --- a/ggml/src/ggml-hip/CMakeLists.txt +++ b/ggml/src/ggml-hip/CMakeLists.txt @@ -81,6 +81,8 @@ ggml_add_backend_library(ggml-hip # TODO: do not use CUDA definitions for HIP if (NOT GGML_BACKEND_DL) target_compile_definitions(ggml PUBLIC GGML_USE_CUDA) + # transcribe.cpp: lets the registry tell HIP from CUDA (reg filter name) + target_compile_definitions(ggml PRIVATE GGML_USE_HIP) endif() add_compile_definitions(GGML_USE_HIP) diff --git a/ggml/src/ggml-musa/CMakeLists.txt b/ggml/src/ggml-musa/CMakeLists.txt index 82b754f41..8681d4189 100644 --- a/ggml/src/ggml-musa/CMakeLists.txt +++ b/ggml/src/ggml-musa/CMakeLists.txt @@ -63,6 +63,8 @@ if (MUSAToolkit_FOUND) # TODO: do not use CUDA definitions for MUSA if (NOT GGML_BACKEND_DL) target_compile_definitions(ggml PUBLIC GGML_USE_CUDA) + # transcribe.cpp: lets the registry tell MUSA from CUDA (reg filter name) + target_compile_definitions(ggml PRIVATE GGML_USE_MUSA) endif() add_compile_definitions(GGML_USE_MUSA) diff --git a/patches/ggml/0002-backend-reg-filter.patch b/patches/ggml/0002-backend-reg-filter.patch index 0ac12f3c5..82fa808db 100644 --- a/patches/ggml/0002-backend-reg-filter.patch +++ b/patches/ggml/0002-backend-reg-filter.patch @@ -21,7 +21,7 @@ index cc3f8cd3..dd98d879 100644 // Backend scheduler // diff --git a/src/ggml-backend-reg.cpp b/src/ggml-backend-reg.cpp -index 1c18b82c..b2a8c193 100644 +index 1c18b82c..085ac303 100644 --- a/src/ggml-backend-reg.cpp +++ b/src/ggml-backend-reg.cpp @@ -3,6 +3,7 @@ @@ -50,12 +50,20 @@ index 1c18b82c..b2a8c193 100644 struct ggml_backend_reg_entry { ggml_backend_reg_t reg; dl_handle_ptr handle; -@@ -118,58 +130,88 @@ struct ggml_backend_registry { +@@ -118,58 +130,96 @@ struct ggml_backend_registry { ggml_backend_registry() { #ifdef GGML_USE_CUDA - register_backend(ggml_backend_cuda_reg()); ++ // HIP and MUSA builds reuse the CUDA backend; filter them under the ++ // name their dynamic module would have. ++#if defined(GGML_USE_HIP) ++ if (reg_allowed("hip")) { ++#elif defined(GGML_USE_MUSA) ++ if (reg_allowed("musa")) { ++#else + if (reg_allowed("cuda")) { ++#endif + register_backend(ggml_backend_cuda_reg()); + } #endif @@ -157,7 +165,7 @@ index 1c18b82c..b2a8c193 100644 #endif } -@@ -478,6 +520,10 @@ static fs::path backend_filename_extension() { +@@ -478,6 +528,10 @@ static fs::path backend_filename_extension() { } static ggml_backend_reg_t ggml_backend_load_best(const char * name, bool silent, const char * user_search_path) { @@ -168,7 +176,7 @@ index 1c18b82c..b2a8c193 100644 // enumerate all the files that match [lib]ggml-name-*.[so|dll] in the search paths const fs::path name_path = fs::u8path(name); const fs::path file_prefix = backend_filename_prefix().native() + name_path.native() + fs::u8path("-").native(); -@@ -599,7 +645,7 @@ void ggml_backend_load_all_from_path(const char * dir_path) { +@@ -599,7 +653,7 @@ void ggml_backend_load_all_from_path(const char * dir_path) { ggml_backend_load_best("cpu", silent, dir_path); // check the environment variable GGML_BACKEND_PATH to load an out-of-tree backend const char * backend_path = std::getenv("GGML_BACKEND_PATH"); @@ -177,3 +185,29 @@ index 1c18b82c..b2a8c193 100644 ggml_backend_load(backend_path); } } +diff --git a/src/ggml-hip/CMakeLists.txt b/src/ggml-hip/CMakeLists.txt +index a6a6b727..6140c29d 100644 +--- a/src/ggml-hip/CMakeLists.txt ++++ b/src/ggml-hip/CMakeLists.txt +@@ -81,6 +81,8 @@ ggml_add_backend_library(ggml-hip + # TODO: do not use CUDA definitions for HIP + if (NOT GGML_BACKEND_DL) + target_compile_definitions(ggml PUBLIC GGML_USE_CUDA) ++ # transcribe.cpp: lets the registry tell HIP from CUDA (reg filter name) ++ target_compile_definitions(ggml PRIVATE GGML_USE_HIP) + endif() + + add_compile_definitions(GGML_USE_HIP) +diff --git a/src/ggml-musa/CMakeLists.txt b/src/ggml-musa/CMakeLists.txt +index 82b754f4..8681d418 100644 +--- a/src/ggml-musa/CMakeLists.txt ++++ b/src/ggml-musa/CMakeLists.txt +@@ -63,6 +63,8 @@ if (MUSAToolkit_FOUND) + # TODO: do not use CUDA definitions for MUSA + if (NOT GGML_BACKEND_DL) + target_compile_definitions(ggml PUBLIC GGML_USE_CUDA) ++ # transcribe.cpp: lets the registry tell MUSA from CUDA (reg filter name) ++ target_compile_definitions(ggml PRIVATE GGML_USE_MUSA) + endif() + + add_compile_definitions(GGML_USE_MUSA) diff --git a/src/transcribe.cpp b/src/transcribe.cpp index 9f3a3306e..6438f5825 100644 --- a/src/transcribe.cpp +++ b/src/transcribe.cpp @@ -1024,9 +1024,17 @@ static std::string path_for_c_api(const std::filesystem::path & path) { // compiled-in backend or opening a backend module, so a backend outside the // mask never runs any code. The first filter call is the moment backends get // registered; from then on the mask is fixed (registrations are permanent). -static std::atomic s_host_backend_mask{ TRANSCRIBE_BACKEND_MASK_ALL }; -static std::atomic s_backend_mask_fixed{ false }; -static std::mutex s_backend_mask_mutex; +// +// The host mask (low 32 bits) and the fixed flag share one atomic so that +// fixing the mask and reading it is a single fetch_or, and setting it is a +// CAS that fails once fixed: a concurrent _ex() and first registration can +// never interleave into "registered under ALL, then _ex(CPU) returned OK". +constexpr uint64_t k_backend_mask_fixed = uint64_t{ 1 } << 32; +static std::atomic s_backend_mask_state{ TRANSCRIBE_BACKEND_MASK_ALL }; + +static uint32_t host_backend_mask() { + return static_cast(s_backend_mask_state.load()); +} static bool ascii_iequals(const char * a, const char * b) { for (; *a != '\0' && *b != '\0'; ++a, ++b) { @@ -1037,6 +1045,17 @@ static bool ascii_iequals(const char * a, const char * b) { return *a == *b; } +// ascii_iequals for a token that is not NUL-terminated. +static bool ascii_iequals_n(const char * a, size_t n, const char * b) { + for (size_t i = 0; i < n; ++i, ++b) { + if (*b == '\0' || + std::tolower(static_cast(a[i])) != std::tolower(static_cast(*b))) { + return false; + } + } + return *b == '\0'; +} + struct BackendMaskName { const char * name; uint32_t bit; @@ -1062,7 +1081,9 @@ static uint32_t module_mask_bit(const char * name) { return TRANSCRIBE_BACKEND_MASK_OTHER; } -// TRANSCRIBE_BACKENDS, parsed once. Unset or empty is inert (ALL). +// TRANSCRIBE_BACKENDS, parsed once. Unset or empty is inert (ALL). Runs +// inside the registry filter, so it must not allocate or throw: tokens are +// matched in place. static uint32_t env_backend_mask() { static const uint32_t mask = [] { static const BackendMaskName k_tokens[] = { @@ -1078,32 +1099,32 @@ static uint32_t env_backend_mask() { if (env == nullptr || env[0] == '\0') { return TRANSCRIBE_BACKEND_MASK_ALL; } - uint32_t m = 0; - std::string tok; + uint32_t m = 0; + const char * tok = env; for (const char * p = env;; ++p) { if (*p != '\0' && *p != ',' && *p != ' ' && *p != '\t') { - tok.push_back(*p); continue; } - if (!tok.empty()) { + const size_t len = static_cast(p - tok); + if (len > 0) { uint32_t bit = 0; for (const auto & e : k_tokens) { - if (ascii_iequals(tok.c_str(), e.name)) { + if (ascii_iequals_n(tok, len, e.name)) { bit = e.bit; } } if (bit == 0) { transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_WARN, - "TRANSCRIBE_BACKENDS: ignoring unknown backend '%s' " + "TRANSCRIBE_BACKENDS: ignoring unknown backend '%.*s' " "(expected cpu, metal, vulkan, cuda, rocm, other, all)", - tok.c_str()); + static_cast(std::min(len, INT_MAX)), tok); } m |= bit; - tok.clear(); } if (*p == '\0') { break; } + tok = p + 1; } return m; }(); @@ -1115,21 +1136,17 @@ static uint32_t effective_backend_mask(uint32_t host_mask) { } // Called by ggml's registry, possibly from inside its function-local static -// constructor: must not touch the registry, must not throw. +// constructor: must not touch the registry. Nothing on this path allocates +// or throws (log_msg formats into a stack buffer and guards the callback). static bool backend_reg_filter(const char * name) noexcept { - try { - s_backend_mask_fixed.store(true); - const uint32_t bit = module_mask_bit(name != nullptr ? name : ""); - const bool allowed = (effective_backend_mask(s_host_backend_mask.load()) & bit) != 0; - if (!allowed) { - transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_DEBUG, - "backend '%s' not registered: excluded by allowed-backend mask", - name != nullptr ? name : "(null)"); - } - return allowed; - } catch (...) { - return false; + const uint32_t host = static_cast(s_backend_mask_state.fetch_or(k_backend_mask_fixed)); + const uint32_t bit = module_mask_bit(name != nullptr ? name : ""); + const bool allowed = (effective_backend_mask(host) & bit) != 0; + if (!allowed) { + transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_DEBUG, "backend '%s' not registered: excluded by allowed-backend mask", + name != nullptr ? name : "(null)"); } + return allowed; } static const bool s_backend_reg_filter_installed = [] { @@ -1138,20 +1155,21 @@ static const bool s_backend_reg_filter_installed = [] { }(); static transcribe_status set_host_backend_mask(uint32_t mask) { - std::lock_guard lock(s_backend_mask_mutex); - if (s_backend_mask_fixed.load()) { - const uint32_t current = effective_backend_mask(s_host_backend_mask.load()); - const uint32_t wanted = effective_backend_mask(mask); - if (current != wanted) { - transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, - "transcribe_init_backends_ex: allowed-backend mask is already fixed at 0x%08x " - "(backends were registered earlier in this process); cannot change it to 0x%08x", - current, wanted); - return TRANSCRIBE_ERR_BACKEND; + uint64_t state = s_backend_mask_state.load(); + while ((state & k_backend_mask_fixed) == 0) { + if (s_backend_mask_state.compare_exchange_weak(state, mask)) { + return TRANSCRIBE_OK; } - return TRANSCRIBE_OK; } - s_host_backend_mask.store(mask); + const uint32_t current = effective_backend_mask(static_cast(state)); + const uint32_t wanted = effective_backend_mask(mask); + if (current != wanted) { + transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, + "transcribe_init_backends_ex: allowed-backend mask is already fixed at 0x%08x " + "(backends were registered earlier in this process); cannot change it to 0x%08x", + current, wanted); + return TRANSCRIBE_ERR_BACKEND; + } return TRANSCRIBE_OK; } @@ -1261,7 +1279,7 @@ static transcribe_status transcribe_init_backends_ex_impl(const struct transcrib if (ggml_backend_dev_count() == 0) { transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "transcribe_init_backends_ex: no compute devices registered (allowed-backend mask 0x%08x)", - effective_backend_mask(s_host_backend_mask.load())); + effective_backend_mask(host_backend_mask())); return TRANSCRIBE_ERR_BACKEND; } return TRANSCRIBE_OK; @@ -3505,7 +3523,7 @@ extern "C" transcribe_status transcribe_init_backends_ex(const struct transcribe extern "C" uint32_t transcribe_allowed_backends(void) { return api_guard_value("transcribe_allowed_backends", static_cast(TRANSCRIBE_BACKEND_MASK_CPU), - [&] { return effective_backend_mask(s_host_backend_mask.load()); }); + [&] { return effective_backend_mask(host_backend_mask()); }); } extern "C" int transcribe_device_count(void) { diff --git a/tests/backend_mask_unit.c b/tests/backend_mask_unit.c index 8e07d7b35..4678ff6d0 100644 --- a/tests/backend_mask_unit.c +++ b/tests/backend_mask_unit.c @@ -26,12 +26,12 @@ static int g_failures = 0; -#define CHECK(cond) \ - do { \ - if (!(cond)) { \ +#define CHECK(cond) \ + do { \ + if (!(cond)) { \ fprintf(stderr, "FAIL %s:%d: %s\n", __FILE__, __LINE__, #cond); \ - ++g_failures; \ - } \ + ++g_failures; \ + } \ } while (0) static int only_cpu_kind_devices(void) { @@ -50,21 +50,28 @@ static int only_cpu_kind_devices(void) { return n > 0; } -/* Linux Vulkan ICDs are libvulkan_.so (radeon, lvp, intel, ...) and - * are only loaded by vkCreateInstance / instance enumeration. Their absence - * from the address space proves the Vulkan backend never initialized. */ +/* Linux Vulkan ICDs are only loaded by vkCreateInstance / instance + * enumeration, so their absence from the address space proves the Vulkan + * backend never initialized. Known ICD libraries: Mesa's libvulkan_ + * (radeon, lvp, intel, ...), NVIDIA's libGLX_nvidia, AMDVLK's amdvlk64/32. + * Other vendors' ICDs are not recognized; run_default() detects that case + * and skips the self-check rather than failing. */ static int vulkan_icd_loaded(void) { #if defined(__linux__) + static const char * const k_icd_libs[] = { "libvulkan_", "libGLX_nvidia", "amdvlk" }; + FILE * f = fopen("/proc/self/maps", "r"); if (f == NULL) { return 0; } char line[4096]; int found = 0; - while (fgets(line, sizeof(line), f) != NULL) { - if (strstr(line, "libvulkan_") != NULL) { - found = 1; - break; + while (!found && fgets(line, sizeof(line), f) != NULL) { + for (size_t i = 0; i < sizeof(k_icd_libs) / sizeof(k_icd_libs[0]); ++i) { + if (strstr(line, k_icd_libs[i]) != NULL) { + found = 1; + break; + } } } fclose(f); @@ -117,10 +124,14 @@ static void run_default(void) { CHECK(transcribe_allowed_backends() == TRANSCRIBE_BACKEND_MASK_ALL); CHECK(transcribe_device_count() > 0); #if defined(__linux__) - /* Probe self-check: the ICD scan must see a Vulkan backend that did - * initialize, or its absence in the other modes proves nothing. */ - if (transcribe_backend_available(TRANSCRIBE_BACKEND_VULKAN)) { - CHECK(vulkan_icd_loaded()); + /* Probe self-check: the ICD scan should see a Vulkan backend that did + * initialize, or its absence in the other modes proves nothing. An + * unrecognized ICD library is a limitation of the probe, not a mask + * failure, so report it and move on. */ + if (transcribe_backend_available(TRANSCRIBE_BACKEND_VULKAN) && !vulkan_icd_loaded()) { + fprintf(stderr, + "note: Vulkan is up but no known ICD library is mapped; the " + "ICD-absence checks in the other modes are not meaningful on this host\n"); } #endif } @@ -134,8 +145,8 @@ static void run_env_cpu(void) { } static void run_late(void) { - const int pre = transcribe_device_count(); - const transcribe_status st = init_with_mask(TRANSCRIBE_BACKEND_MASK_CPU); + const int pre = transcribe_device_count(); + const transcribe_status st = init_with_mask(TRANSCRIBE_BACKEND_MASK_CPU); if (pre > 0) { /* Static build: the query already registered everything under ALL, * so narrowing now is refused. */ From aa892e7b972982c763d851ca57cbf11dd5acaceb Mon Sep 17 00:00:00 2001 From: CJ Pais Date: Sun, 4 Oct 2026 08:46:57 +0800 Subject: [PATCH 7/8] slim comments --- bindings/rust/transcribe-cpp/src/backend.rs | 39 +++------- .../rust/transcribe-cpp/tests/backend_mask.rs | 5 +- ggml/include/ggml-backend.h | 12 +-- ggml/src/ggml-backend-reg.cpp | 3 +- ggml/src/ggml-hip/CMakeLists.txt | 2 +- ggml/src/ggml-musa/CMakeLists.txt | 2 +- include/transcribe.h | 75 +++++-------------- patches/ggml/0002-backend-reg-filter.patch | 35 ++++----- src/transcribe.cpp | 28 ++----- 9 files changed, 65 insertions(+), 136 deletions(-) diff --git a/bindings/rust/transcribe-cpp/src/backend.rs b/bindings/rust/transcribe-cpp/src/backend.rs index aa176ac56..c18edb930 100644 --- a/bindings/rust/transcribe-cpp/src/backend.rs +++ b/bindings/rust/transcribe-cpp/src/backend.rs @@ -157,16 +157,10 @@ pub fn init_backends_default() -> Result<()> { check(status, "init_backends_default") } -/// A set of backend kinds allowed to register in this process. -/// -/// Registering a GPU backend runs driver code (Vulkan instance creation loads -/// every installed driver; Metal and CUDA initialize theirs), so a broken -/// driver can crash or hang the process before any model loads. A backend -/// outside the mask is never registered: its module is never opened and its -/// registration function never runs. The CPU backend is always allowed. -/// -/// The `TRANSCRIBE_BACKENDS` environment variable (e.g. `cpu` or -/// `cpu,vulkan`) can only narrow the mask further. +/// Backend kinds allowed to register in this process. A backend outside the +/// mask never runs any code, so a broken GPU driver can be kept out of a +/// worker entirely. CPU is always allowed; `TRANSCRIBE_BACKENDS` can only +/// narrow the mask. #[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] #[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))] pub struct BackendMask(u32); @@ -187,11 +181,8 @@ impl BackendMask { /// Everything. The default. pub const ALL: BackendMask = BackendMask(sys::TRANSCRIBE_BACKEND_MASK_ALL); - /// The smallest mask that can satisfy a model-load `backend` request: - /// [`Backend::Auto`] needs everything, [`Backend::Cpu`] / - /// [`Backend::CpuAccel`] only the CPU, and an explicit GPU backend only - /// itself (plus the implied CPU). A worker process that serves exactly - /// one request kind passes this to [`init_backends_with`]. + /// The smallest mask that can serve a model-load `backend` request + /// ([`Backend::Auto`] needs everything). pub const fn for_backend(backend: Backend) -> BackendMask { match backend { Backend::Auto => BackendMask::ALL, @@ -243,15 +234,10 @@ impl std::ops::BitOrAssign for BackendMask { } } -/// [`init_backends`] / [`init_backends_default`] with an allowed-backend mask. -/// `dir` is the backend module directory, or `None` for the package-local -/// default. -/// -/// The mask is fixed the first time backends register in this process (this -/// call, or in static builds the first device query or model load). Call it -/// first, once per process: a later call asking for a different effective -/// mask errors with [`crate::Error::Backend`]. Registration is permanent, so a -/// worker that must avoid a backend needs a fresh process. +/// [`init_backends`] / [`init_backends_default`] (`dir` = `None`) with an +/// allowed-backend mask. The mask is fixed at first backend registration; +/// call this first, once per process. A later call with a different +/// effective mask returns [`crate::Error::Backend`]. /// /// ```no_run /// use transcribe_cpp::{init_backends_with, Backend, BackendMask}; @@ -272,9 +258,8 @@ pub fn init_backends_with(dir: Option>, allowed: BackendMask) - check(status, "init_backends_with") } -/// The effective allowed-backend mask: the mask passed to -/// [`init_backends_with`] (everything until then), narrowed by -/// `TRANSCRIBE_BACKENDS`, with the CPU always included. +/// The effective mask: [`init_backends_with`]'s (ALL until then) narrowed +/// by `TRANSCRIBE_BACKENDS`. pub fn allowed_backends() -> BackendMask { BackendMask(unsafe { sys::transcribe_allowed_backends() }) } diff --git a/bindings/rust/transcribe-cpp/tests/backend_mask.rs b/bindings/rust/transcribe-cpp/tests/backend_mask.rs index c4e778040..0f7adc871 100644 --- a/bindings/rust/transcribe-cpp/tests/backend_mask.rs +++ b/bindings/rust/transcribe-cpp/tests/backend_mask.rs @@ -11,7 +11,10 @@ use transcribe_cpp::{ fn for_backend_is_minimal() { assert_eq!(BackendMask::for_backend(Backend::Auto), BackendMask::ALL); assert_eq!(BackendMask::for_backend(Backend::Cpu), BackendMask::CPU); - assert_eq!(BackendMask::for_backend(Backend::CpuAccel), BackendMask::CPU); + assert_eq!( + BackendMask::for_backend(Backend::CpuAccel), + BackendMask::CPU + ); assert_eq!( BackendMask::for_backend(Backend::Vulkan), BackendMask::VULKAN | BackendMask::CPU diff --git a/ggml/include/ggml-backend.h b/ggml/include/ggml-backend.h index dd98d879c..764ffe849 100644 --- a/ggml/include/ggml-backend.h +++ b/ggml/include/ggml-backend.h @@ -259,14 +259,10 @@ extern "C" { GGML_API void ggml_backend_load_all(void); GGML_API void ggml_backend_load_all_from_path(const char * dir_path); - // Registration filter (transcribe.cpp downstream patch). When set, the - // registry consults it before registering each compiled-in backend and - // before opening any dynamic backend module, keyed by the lower-case - // module name ("cpu", "blas", "metal", "vulkan", "cuda", "hip", ...; - // "external" for GGML_BACKEND_PATH). A rejected backend's code never runs: - // no reg function call, no dlopen. Explicit ggml_backend_load(path) calls - // are not filtered. Install before the first registry access so the - // compiled-in backends see it. NULL (the default) allows everything. + // Registration filter (transcribe.cpp patch): consulted with the module + // name ("cpu", "vulkan", "hip", ...; "external" for GGML_BACKEND_PATH) + // before registering a compiled-in backend or opening a module. Install + // before first registry access. ggml_backend_load(path) is not filtered. typedef bool (*ggml_backend_reg_filter_t)(const char * name); GGML_API void ggml_backend_set_reg_filter(ggml_backend_reg_filter_t filter); diff --git a/ggml/src/ggml-backend-reg.cpp b/ggml/src/ggml-backend-reg.cpp index 085ac3034..7c076862c 100644 --- a/ggml/src/ggml-backend-reg.cpp +++ b/ggml/src/ggml-backend-reg.cpp @@ -130,8 +130,7 @@ struct ggml_backend_registry { ggml_backend_registry() { #ifdef GGML_USE_CUDA - // HIP and MUSA builds reuse the CUDA backend; filter them under the - // name their dynamic module would have. + // HIP/MUSA reuse the CUDA backend; filter under their module name. #if defined(GGML_USE_HIP) if (reg_allowed("hip")) { #elif defined(GGML_USE_MUSA) diff --git a/ggml/src/ggml-hip/CMakeLists.txt b/ggml/src/ggml-hip/CMakeLists.txt index 6140c29dd..9b963dd5f 100644 --- a/ggml/src/ggml-hip/CMakeLists.txt +++ b/ggml/src/ggml-hip/CMakeLists.txt @@ -81,7 +81,7 @@ ggml_add_backend_library(ggml-hip # TODO: do not use CUDA definitions for HIP if (NOT GGML_BACKEND_DL) target_compile_definitions(ggml PUBLIC GGML_USE_CUDA) - # transcribe.cpp: lets the registry tell HIP from CUDA (reg filter name) + # transcribe.cpp: reg filter name target_compile_definitions(ggml PRIVATE GGML_USE_HIP) endif() diff --git a/ggml/src/ggml-musa/CMakeLists.txt b/ggml/src/ggml-musa/CMakeLists.txt index 8681d4189..6cb178ebc 100644 --- a/ggml/src/ggml-musa/CMakeLists.txt +++ b/ggml/src/ggml-musa/CMakeLists.txt @@ -63,7 +63,7 @@ if (MUSAToolkit_FOUND) # TODO: do not use CUDA definitions for MUSA if (NOT GGML_BACKEND_DL) target_compile_definitions(ggml PUBLIC GGML_USE_CUDA) - # transcribe.cpp: lets the registry tell MUSA from CUDA (reg filter name) + # transcribe.cpp: reg filter name target_compile_definitions(ggml PRIVATE GGML_USE_MUSA) endif() diff --git a/include/transcribe.h b/include/transcribe.h index 865cfb110..df3772339 100644 --- a/include/transcribe.h +++ b/include/transcribe.h @@ -826,44 +826,18 @@ TRANSCRIBE_API transcribe_status transcribe_init_backends(const char * artifact_ TRANSCRIBE_API transcribe_status transcribe_init_backends_default(void); /* - * Restricting which backends may initialize. - * - * Registering a GPU backend runs driver code: Vulkan creates an instance - * (loading every installed ICD), Metal opens the system device, CUDA - * initializes the driver. A broken driver can crash or hang the process - * right there, before any model is loaded. A host that isolates inference - * in a worker process can recover from that only if the replacement worker - * never runs the failing backend's code at all — hiding its devices after - * registration is too late. - * - * transcribe_init_backends_ex() takes an allow-mask of backend kinds. A - * backend outside the mask is never registered: its module is never opened - * (dynamic-backend builds) and its registration function is never called - * (static builds). The CPU backend is always allowed, so a mask of 0 or - * TRANSCRIBE_BACKEND_MASK_CPU means "CPU only". - * - * CPU the CPU backend plus host-memory accelerators (BLAS, ZenDNN) - * METAL Apple Metal - * VULKAN Vulkan - * CUDA NVIDIA CUDA - * ROCM AMD ROCm / HIP - * OTHER every backend without a dedicated bit in the running library - * (SYCL, OpenCL, RPC, ..., and an out-of-tree module named by - * GGML_BACKEND_PATH) - * - * The TRANSCRIBE_BACKENDS environment variable can only narrow the mask - * further: a comma-separated list of cpu, metal, vulkan, cuda, rocm, other, - * all (case-insensitive; e.g. TRANSCRIBE_BACKENDS=cpu forces CPU-only - * regardless of what the host passes). Unset or empty means "all". Unknown - * names are logged and ignored. It applies to every way backends get - * registered, including hosts that never call this function. - * - * The mask is FIXED the first time the library registers backends — that - * is, the first transcribe_init_backends*() call or, in static builds, the - * first call that enumerates devices or loads a model. Backends cannot be - * unregistered, so a later call asking for a different effective mask - * returns TRANSCRIBE_ERR_BACKEND without changing anything. Call this once, - * first, per process. + * Allowed-backend mask. Registering a GPU backend runs driver code, so a + * broken driver can crash the process before any model loads. A backend + * outside the mask is never registered: its module is never opened and its + * registration function never runs. CPU (incl. BLAS/ZenDNN) is always + * allowed; OTHER covers backends without a bit (SYCL, OpenCL, RPC, ...). + * + * TRANSCRIBE_BACKENDS=cpu,vulkan,... (also metal, cuda, rocm, other, all) + * can only narrow the mask, and applies even if _ex() is never called. + * + * The mask is fixed at first backend registration (first init call, or in + * static builds the first device query / model load). A later call with a + * different effective mask returns TRANSCRIBE_ERR_BACKEND. Call once, first. */ #define TRANSCRIBE_BACKEND_MASK_CPU (1u << 0) #define TRANSCRIBE_BACKEND_MASK_METAL (1u << 1) @@ -875,32 +849,21 @@ TRANSCRIBE_API transcribe_status transcribe_init_backends_default(void); struct transcribe_backend_init_params { uint64_t struct_size; /* sizeof(*this); set by _init() */ - const char * artifact_dir; /* NULL: package-local default, as - transcribe_init_backends_default() */ - uint32_t allowed_backends; /* TRANSCRIBE_BACKEND_MASK_* bits; - _init() sets ..._MASK_ALL */ + const char * artifact_dir; /* NULL: package-local default */ + uint32_t allowed_backends; /* TRANSCRIBE_BACKEND_MASK_*; default ALL */ }; TRANSCRIBE_API void transcribe_backend_init_params_init(struct transcribe_backend_init_params * p); /* - * Fix the allowed-backend mask (see above), then load backend modules as - * transcribe_init_backends(artifact_dir) or, with artifact_dir NULL, - * transcribe_init_backends_default() would. NULL params means all defaults. - * - * Returns the statuses of those calls, plus: - * TRANSCRIBE_ERR_BAD_STRUCT_SIZE params fails the struct-size check. - * TRANSCRIBE_ERR_BACKEND the mask was already fixed to a different - * effective value, or no compute device is - * registered afterwards. + * Fix the mask, then behave as transcribe_init_backends(artifact_dir), or + * _default() when artifact_dir is NULL. NULL params means all defaults. + * Also returns TRANSCRIBE_ERR_BACKEND if the mask was already fixed to a + * different value or no device is registered afterwards. */ TRANSCRIBE_API transcribe_status transcribe_init_backends_ex(const struct transcribe_backend_init_params * params); -/* - * The effective allowed-backend mask: the host's mask (ALL until - * transcribe_init_backends_ex() sets one) narrowed by TRANSCRIBE_BACKENDS, - * with the CPU bit always set. - */ +/* The effective mask: the host's (ALL until _ex()) narrowed by the env. */ TRANSCRIBE_API uint32_t transcribe_allowed_backends(void); /* diff --git a/patches/ggml/0002-backend-reg-filter.patch b/patches/ggml/0002-backend-reg-filter.patch index 82fa808db..b52ca4df9 100644 --- a/patches/ggml/0002-backend-reg-filter.patch +++ b/patches/ggml/0002-backend-reg-filter.patch @@ -1,19 +1,15 @@ diff --git a/include/ggml-backend.h b/include/ggml-backend.h -index cc3f8cd3..dd98d879 100644 +index cc3f8cd3..764ffe84 100644 --- a/include/ggml-backend.h +++ b/include/ggml-backend.h -@@ -259,6 +259,17 @@ extern "C" { +@@ -259,6 +259,13 @@ extern "C" { GGML_API void ggml_backend_load_all(void); GGML_API void ggml_backend_load_all_from_path(const char * dir_path); -+ // Registration filter (transcribe.cpp downstream patch). When set, the -+ // registry consults it before registering each compiled-in backend and -+ // before opening any dynamic backend module, keyed by the lower-case -+ // module name ("cpu", "blas", "metal", "vulkan", "cuda", "hip", ...; -+ // "external" for GGML_BACKEND_PATH). A rejected backend's code never runs: -+ // no reg function call, no dlopen. Explicit ggml_backend_load(path) calls -+ // are not filtered. Install before the first registry access so the -+ // compiled-in backends see it. NULL (the default) allows everything. ++ // Registration filter (transcribe.cpp patch): consulted with the module ++ // name ("cpu", "vulkan", "hip", ...; "external" for GGML_BACKEND_PATH) ++ // before registering a compiled-in backend or opening a module. Install ++ // before first registry access. ggml_backend_load(path) is not filtered. + typedef bool (*ggml_backend_reg_filter_t)(const char * name); + GGML_API void ggml_backend_set_reg_filter(ggml_backend_reg_filter_t filter); + @@ -21,7 +17,7 @@ index cc3f8cd3..dd98d879 100644 // Backend scheduler // diff --git a/src/ggml-backend-reg.cpp b/src/ggml-backend-reg.cpp -index 1c18b82c..085ac303 100644 +index 1c18b82c..7c076862 100644 --- a/src/ggml-backend-reg.cpp +++ b/src/ggml-backend-reg.cpp @@ -3,6 +3,7 @@ @@ -50,13 +46,12 @@ index 1c18b82c..085ac303 100644 struct ggml_backend_reg_entry { ggml_backend_reg_t reg; dl_handle_ptr handle; -@@ -118,58 +130,96 @@ struct ggml_backend_registry { +@@ -118,58 +130,95 @@ struct ggml_backend_registry { ggml_backend_registry() { #ifdef GGML_USE_CUDA - register_backend(ggml_backend_cuda_reg()); -+ // HIP and MUSA builds reuse the CUDA backend; filter them under the -+ // name their dynamic module would have. ++ // HIP/MUSA reuse the CUDA backend; filter under their module name. +#if defined(GGML_USE_HIP) + if (reg_allowed("hip")) { +#elif defined(GGML_USE_MUSA) @@ -165,7 +160,7 @@ index 1c18b82c..085ac303 100644 #endif } -@@ -478,6 +528,10 @@ static fs::path backend_filename_extension() { +@@ -478,6 +527,10 @@ static fs::path backend_filename_extension() { } static ggml_backend_reg_t ggml_backend_load_best(const char * name, bool silent, const char * user_search_path) { @@ -176,7 +171,7 @@ index 1c18b82c..085ac303 100644 // enumerate all the files that match [lib]ggml-name-*.[so|dll] in the search paths const fs::path name_path = fs::u8path(name); const fs::path file_prefix = backend_filename_prefix().native() + name_path.native() + fs::u8path("-").native(); -@@ -599,7 +653,7 @@ void ggml_backend_load_all_from_path(const char * dir_path) { +@@ -599,7 +652,7 @@ void ggml_backend_load_all_from_path(const char * dir_path) { ggml_backend_load_best("cpu", silent, dir_path); // check the environment variable GGML_BACKEND_PATH to load an out-of-tree backend const char * backend_path = std::getenv("GGML_BACKEND_PATH"); @@ -186,27 +181,27 @@ index 1c18b82c..085ac303 100644 } } diff --git a/src/ggml-hip/CMakeLists.txt b/src/ggml-hip/CMakeLists.txt -index a6a6b727..6140c29d 100644 +index a6a6b727..9b963dd5 100644 --- a/src/ggml-hip/CMakeLists.txt +++ b/src/ggml-hip/CMakeLists.txt @@ -81,6 +81,8 @@ ggml_add_backend_library(ggml-hip # TODO: do not use CUDA definitions for HIP if (NOT GGML_BACKEND_DL) target_compile_definitions(ggml PUBLIC GGML_USE_CUDA) -+ # transcribe.cpp: lets the registry tell HIP from CUDA (reg filter name) ++ # transcribe.cpp: reg filter name + target_compile_definitions(ggml PRIVATE GGML_USE_HIP) endif() add_compile_definitions(GGML_USE_HIP) diff --git a/src/ggml-musa/CMakeLists.txt b/src/ggml-musa/CMakeLists.txt -index 82b754f4..8681d418 100644 +index 82b754f4..6cb178eb 100644 --- a/src/ggml-musa/CMakeLists.txt +++ b/src/ggml-musa/CMakeLists.txt @@ -63,6 +63,8 @@ if (MUSAToolkit_FOUND) # TODO: do not use CUDA definitions for MUSA if (NOT GGML_BACKEND_DL) target_compile_definitions(ggml PUBLIC GGML_USE_CUDA) -+ # transcribe.cpp: lets the registry tell MUSA from CUDA (reg filter name) ++ # transcribe.cpp: reg filter name + target_compile_definitions(ggml PRIVATE GGML_USE_MUSA) endif() diff --git a/src/transcribe.cpp b/src/transcribe.cpp index 6438f5825..28f0bd82c 100644 --- a/src/transcribe.cpp +++ b/src/transcribe.cpp @@ -1019,16 +1019,9 @@ static std::string path_for_c_api(const std::filesystem::path & path) { } #endif -// Allowed-backend mask. ggml consults backend_reg_filter (installed at static -// init, before anything can touch ggml's registry) before registering a -// compiled-in backend or opening a backend module, so a backend outside the -// mask never runs any code. The first filter call is the moment backends get -// registered; from then on the mask is fixed (registrations are permanent). -// -// The host mask (low 32 bits) and the fixed flag share one atomic so that -// fixing the mask and reading it is a single fetch_or, and setting it is a -// CAS that fails once fixed: a concurrent _ex() and first registration can -// never interleave into "registered under ALL, then _ex(CPU) returned OK". +// Allowed-backend mask, enforced by backend_reg_filter (installed at static +// init). The first filter call fixes the mask. Mask (low 32 bits) and fixed +// flag share one atomic so _ex() cannot race the first registration. constexpr uint64_t k_backend_mask_fixed = uint64_t{ 1 } << 32; static std::atomic s_backend_mask_state{ TRANSCRIBE_BACKEND_MASK_ALL }; @@ -1061,8 +1054,7 @@ struct BackendMaskName { uint32_t bit; }; -// ggml module names (the [lib]ggml- stem) -> mask bit. Anything not -// listed, including "external" (GGML_BACKEND_PATH), is OTHER. +// ggml module name -> mask bit; unlisted (incl. "external") is OTHER. static uint32_t module_mask_bit(const char * name) { static const BackendMaskName k_modules[] = { { "cpu", TRANSCRIBE_BACKEND_MASK_CPU }, @@ -1081,9 +1073,8 @@ static uint32_t module_mask_bit(const char * name) { return TRANSCRIBE_BACKEND_MASK_OTHER; } -// TRANSCRIBE_BACKENDS, parsed once. Unset or empty is inert (ALL). Runs -// inside the registry filter, so it must not allocate or throw: tokens are -// matched in place. +// TRANSCRIBE_BACKENDS, parsed once; unset/empty is ALL. Runs inside the +// registry filter, so it must not allocate or throw. static uint32_t env_backend_mask() { static const uint32_t mask = [] { static const BackendMaskName k_tokens[] = { @@ -1135,9 +1126,7 @@ static uint32_t effective_backend_mask(uint32_t host_mask) { return (host_mask & env_backend_mask()) | TRANSCRIBE_BACKEND_MASK_CPU; } -// Called by ggml's registry, possibly from inside its function-local static -// constructor: must not touch the registry. Nothing on this path allocates -// or throws (log_msg formats into a stack buffer and guards the callback). +// Called from inside ggml's registry: must not touch the registry or throw. static bool backend_reg_filter(const char * name) noexcept { const uint32_t host = static_cast(s_backend_mask_state.fetch_or(k_backend_mask_fixed)); const uint32_t bit = module_mask_bit(name != nullptr ? name : ""); @@ -1274,8 +1263,7 @@ static transcribe_status transcribe_init_backends_ex_impl(const struct transcrib if (st != TRANSCRIBE_OK) { return st; } - // Static builds register compiled-in backends lazily on first registry - // access; force it here so the mask is fixed by this call, as documented. + // Static builds register lazily; force it so this call fixes the mask. if (ggml_backend_dev_count() == 0) { transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "transcribe_init_backends_ex: no compute devices registered (allowed-backend mask 0x%08x)", From 11979940dea9a59d65e2145f193ceb7b2d690178 Mon Sep 17 00:00:00 2001 From: CJ Pais Date: Sun, 4 Oct 2026 10:04:16 +0800 Subject: [PATCH 8/8] backend mask: literal bits, ABI id, callback-only filter log, loud env typo, env docs --- .../python/src/transcribe_cpp/_generated.py | 10 ++++- bindings/rust/sys/src/transcribe_sys.rs | 5 ++- bindings/rust/transcribe-cpp/src/types.rs | 2 + .../rust/transcribe-cpp/tests/no_model.rs | 1 + .../swift/Sources/TranscribeCpp/ABIHash.swift | 2 +- bindings/typescript/src/_generated.ts | 10 ++++- docs/environment-variables.md | 1 + include/transcribe.abihash | 2 +- include/transcribe.h | 43 ++++++++++--------- src/transcribe.cpp | 42 +++++++++++++----- 10 files changed, 81 insertions(+), 37 deletions(-) diff --git a/bindings/python/src/transcribe_cpp/_generated.py b/bindings/python/src/transcribe_cpp/_generated.py index 29b3b2435..bf84e145c 100644 --- a/bindings/python/src/transcribe_cpp/_generated.py +++ b/bindings/python/src/transcribe_cpp/_generated.py @@ -13,7 +13,7 @@ # Stable digest of the ABI surface below (structs, enums, macros, layout, # prototypes). A native provider package echoes this back so the API # package can reject an ABI-mismatched provider before dlopen. -PUBLIC_HEADER_HASH = "8c53a55291ec1597" +PUBLIC_HEADER_HASH = "57b1af43650f195d" # === enum constants === TRANSCRIBE_OK = 0 @@ -51,6 +51,7 @@ TRANSCRIBE_ABI_EXT = 12 TRANSCRIBE_ABI_DEVICE_INFO = 13 TRANSCRIBE_ABI_SPEAKER_SEGMENT = 14 +TRANSCRIBE_ABI_BACKEND_INIT_PARAMS = 15 TRANSCRIBE_LOG_LEVEL_NONE = 0 TRANSCRIBE_LOG_LEVEL_INFO = 1 TRANSCRIBE_LOG_LEVEL_WARN = 2 @@ -117,6 +118,12 @@ # === macro constants (integer object-like macros) === TRANSCRIBE_BACKEND_MASK_ALL = 4294967295 +TRANSCRIBE_BACKEND_MASK_CPU = 1 +TRANSCRIBE_BACKEND_MASK_CUDA = 8 +TRANSCRIBE_BACKEND_MASK_METAL = 2 +TRANSCRIBE_BACKEND_MASK_OTHER = 2147483648 +TRANSCRIBE_BACKEND_MASK_ROCM = 16 +TRANSCRIBE_BACKEND_MASK_VULKAN = 4 TRANSCRIBE_EXT_KIND_MOONSHINE_STREAMING_STREAM = 1414746957 TRANSCRIBE_EXT_KIND_PARAKEET_BUFFERED_STREAM = 1396853584 TRANSCRIBE_EXT_KIND_PARAKEET_STREAM = 1414744912 @@ -200,6 +207,7 @@ class transcribe_whisper_chunk_trace(_c.Structure): # transcribe_abi_struct id per struct (for the native size/align check). ABI_STRUCT_IDS = { 'transcribe_ext': 12, + 'transcribe_backend_init_params': 15, 'transcribe_device_info': 13, 'transcribe_model_load_params': 0, 'transcribe_session_params': 1, diff --git a/bindings/rust/sys/src/transcribe_sys.rs b/bindings/rust/sys/src/transcribe_sys.rs index 7eb810849..f679cb8ff 100644 --- a/bindings/rust/sys/src/transcribe_sys.rs +++ b/bindings/rust/sys/src/transcribe_sys.rs @@ -1,11 +1,11 @@ // @generated by `cargo xtask bindgen` from include/transcribe/extensions.h // DO NOT EDIT BY HAND. Regenerate: `cargo xtask bindgen`. -// Pinned to include/transcribe.abihash = 8c53a55291ec1597 +// Pinned to include/transcribe.abihash = 57b1af43650f195d /// The public-ABI digest these bindings were generated against /// (sha256/16 over the normalized FFI surface). The load-time version /// gate and the CI drift check both anchor on this value. -pub const PUBLIC_HEADER_HASH: &str = "8c53a55291ec1597"; +pub const PUBLIC_HEADER_HASH: &str = "57b1af43650f195d"; /* automatically generated by rust-bindgen 0.72.1 */ @@ -73,6 +73,7 @@ impl transcribe_abi_struct { pub const TRANSCRIBE_ABI_EXT: transcribe_abi_struct = transcribe_abi_struct(12); pub const TRANSCRIBE_ABI_DEVICE_INFO: transcribe_abi_struct = transcribe_abi_struct(13); pub const TRANSCRIBE_ABI_SPEAKER_SEGMENT: transcribe_abi_struct = transcribe_abi_struct(14); + pub const TRANSCRIBE_ABI_BACKEND_INIT_PARAMS: transcribe_abi_struct = transcribe_abi_struct(15); } #[repr(transparent)] #[derive(Debug, Copy, Clone, Hash, PartialEq, Eq)] diff --git a/bindings/rust/transcribe-cpp/src/types.rs b/bindings/rust/transcribe-cpp/src/types.rs index af9d95000..4c780c6bd 100644 --- a/bindings/rust/transcribe-cpp/src/types.rs +++ b/bindings/rust/transcribe-cpp/src/types.rs @@ -319,6 +319,7 @@ pub enum AbiStruct { Ext, DeviceInfo, SpeakerSegment, + BackendInitParams, } impl AbiStruct { @@ -340,6 +341,7 @@ impl AbiStruct { AbiStruct::Ext => A::TRANSCRIBE_ABI_EXT, AbiStruct::DeviceInfo => A::TRANSCRIBE_ABI_DEVICE_INFO, AbiStruct::SpeakerSegment => A::TRANSCRIBE_ABI_SPEAKER_SEGMENT, + AbiStruct::BackendInitParams => A::TRANSCRIBE_ABI_BACKEND_INIT_PARAMS, } } } diff --git a/bindings/rust/transcribe-cpp/tests/no_model.rs b/bindings/rust/transcribe-cpp/tests/no_model.rs index 89a750be5..9aa2c1e92 100644 --- a/bindings/rust/transcribe-cpp/tests/no_model.rs +++ b/bindings/rust/transcribe-cpp/tests/no_model.rs @@ -34,6 +34,7 @@ fn abi_struct_sizes_are_live() { AbiStruct::Segment, AbiStruct::SpeakerSegment, AbiStruct::SessionLimits, + AbiStruct::BackendInitParams, ] { assert!(abi_struct_size(which) > 0, "{which:?} reported size 0"); } diff --git a/bindings/swift/Sources/TranscribeCpp/ABIHash.swift b/bindings/swift/Sources/TranscribeCpp/ABIHash.swift index 34f41a228..d2c8808fe 100644 --- a/bindings/swift/Sources/TranscribeCpp/ABIHash.swift +++ b/bindings/swift/Sources/TranscribeCpp/ABIHash.swift @@ -13,7 +13,7 @@ import CTranscribe extension Transcribe { /// sha256/16 of the normalized public FFI surface, pinned to the value in /// include/transcribe.abihash at the time this binding was last reviewed. - public static let pinnedHeaderHash = "8c53a55291ec1597" + public static let pinnedHeaderHash = "57b1af43650f195d" /// The public-ABI digest this binding was reviewed against (16 hex chars). public static func headerHash() -> String { pinnedHeaderHash } diff --git a/bindings/typescript/src/_generated.ts b/bindings/typescript/src/_generated.ts index 52e980540..e5e5be8aa 100644 --- a/bindings/typescript/src/_generated.ts +++ b/bindings/typescript/src/_generated.ts @@ -11,7 +11,7 @@ // Stable digest of the ABI surface (structs, enums, macros, layout, // prototypes), computed by the Python oracle and pinned here so a header // ABI change turns this binding's drift check red for conscious review. -export const PUBLIC_HEADER_HASH = "8c53a55291ec1597"; +export const PUBLIC_HEADER_HASH = "57b1af43650f195d"; // === enum constants === export const TRANSCRIBE_OK = 0; @@ -49,6 +49,7 @@ export const TRANSCRIBE_ABI_SESSION_LIMITS = 11; export const TRANSCRIBE_ABI_EXT = 12; export const TRANSCRIBE_ABI_DEVICE_INFO = 13; export const TRANSCRIBE_ABI_SPEAKER_SEGMENT = 14; +export const TRANSCRIBE_ABI_BACKEND_INIT_PARAMS = 15; export const TRANSCRIBE_LOG_LEVEL_NONE = 0; export const TRANSCRIBE_LOG_LEVEL_INFO = 1; export const TRANSCRIBE_LOG_LEVEL_WARN = 2; @@ -115,6 +116,12 @@ export const TRANSCRIBE_WHISPER_PROMPT_ALL_SEGMENTS = 1; // === macro constants (integer object-like macros) === export const TRANSCRIBE_BACKEND_MASK_ALL = 4294967295; +export const TRANSCRIBE_BACKEND_MASK_CPU = 1; +export const TRANSCRIBE_BACKEND_MASK_CUDA = 8; +export const TRANSCRIBE_BACKEND_MASK_METAL = 2; +export const TRANSCRIBE_BACKEND_MASK_OTHER = 2147483648; +export const TRANSCRIBE_BACKEND_MASK_ROCM = 16; +export const TRANSCRIBE_BACKEND_MASK_VULKAN = 4; export const TRANSCRIBE_EXT_KIND_MOONSHINE_STREAMING_STREAM = 1414746957; export const TRANSCRIBE_EXT_KIND_PARAKEET_BUFFERED_STREAM = 1396853584; export const TRANSCRIBE_EXT_KIND_PARAKEET_STREAM = 1414744912; @@ -151,6 +158,7 @@ export const STRUCT_LAYOUT: Record = { export const ABI_STRUCT_IDS: Record = { 'transcribe_ext': 12, + 'transcribe_backend_init_params': 15, 'transcribe_device_info': 13, 'transcribe_model_load_params': 0, 'transcribe_session_params': 1, diff --git a/docs/environment-variables.md b/docs/environment-variables.md index 848a6257a..f8147112b 100644 --- a/docs/environment-variables.md +++ b/docs/environment-variables.md @@ -24,6 +24,7 @@ of tests. | Variable | Effect | | --- | --- | +| `TRANSCRIBE_BACKENDS=` | Restrict which backends may register: a comma-separated list of `cpu`, `metal`, `vulkan`, `cuda`, `rocm`, `other`, `all` (case-insensitive). An excluded backend never runs any code (no module load, no driver init). It can only narrow the mask a host passes to `transcribe_init_backends_ex()`, applies even if the host never calls it, and CPU is always kept. Unknown names are dropped and logged as an error naming what is still allowed, so a typo narrows rather than widens (`vulcan` means CPU-only). Unset or empty means `all`. Read once per process. | | `TRANSCRIBE_NO_FLASH` | Disable flash attention on encoder and decoder (forces the manual F32 path). | | `TRANSCRIBE_FORCE_FLASH` | Force flash attention on. Wins over `TRANSCRIBE_NO_FLASH` if both are set. | | `TRANSCRIBE_CONV_DIRECT_DW` / `TRANSCRIBE_CONV_NO_DIRECT_DW` | Force the depthwise-conv dispatch to the direct `conv_2d_dw` path / the im2col path, overriding the per-family backend default. | diff --git a/include/transcribe.abihash b/include/transcribe.abihash index 0083d8b24..4d382de59 100644 --- a/include/transcribe.abihash +++ b/include/transcribe.abihash @@ -1 +1 @@ -8c53a55291ec1597 +57b1af43650f195d diff --git a/include/transcribe.h b/include/transcribe.h index df3772339..6be8e8ca3 100644 --- a/include/transcribe.h +++ b/include/transcribe.h @@ -369,21 +369,22 @@ TRANSCRIBE_API const char * transcribe_version_commit(void); * is append-only; do not renumber existing values. */ typedef enum { - TRANSCRIBE_ABI_MODEL_LOAD_PARAMS = 0, - TRANSCRIBE_ABI_SESSION_PARAMS = 1, - TRANSCRIBE_ABI_RUN_PARAMS = 2, - TRANSCRIBE_ABI_STREAM_PARAMS = 3, - TRANSCRIBE_ABI_CAPABILITIES = 4, - TRANSCRIBE_ABI_TIMINGS = 5, - TRANSCRIBE_ABI_SEGMENT = 6, - TRANSCRIBE_ABI_WORD = 7, - TRANSCRIBE_ABI_TOKEN = 8, - TRANSCRIBE_ABI_STREAM_UPDATE = 9, - TRANSCRIBE_ABI_STREAM_TEXT = 10, - TRANSCRIBE_ABI_SESSION_LIMITS = 11, - TRANSCRIBE_ABI_EXT = 12, - TRANSCRIBE_ABI_DEVICE_INFO = 13, - TRANSCRIBE_ABI_SPEAKER_SEGMENT = 14, + TRANSCRIBE_ABI_MODEL_LOAD_PARAMS = 0, + TRANSCRIBE_ABI_SESSION_PARAMS = 1, + TRANSCRIBE_ABI_RUN_PARAMS = 2, + TRANSCRIBE_ABI_STREAM_PARAMS = 3, + TRANSCRIBE_ABI_CAPABILITIES = 4, + TRANSCRIBE_ABI_TIMINGS = 5, + TRANSCRIBE_ABI_SEGMENT = 6, + TRANSCRIBE_ABI_WORD = 7, + TRANSCRIBE_ABI_TOKEN = 8, + TRANSCRIBE_ABI_STREAM_UPDATE = 9, + TRANSCRIBE_ABI_STREAM_TEXT = 10, + TRANSCRIBE_ABI_SESSION_LIMITS = 11, + TRANSCRIBE_ABI_EXT = 12, + TRANSCRIBE_ABI_DEVICE_INFO = 13, + TRANSCRIBE_ABI_SPEAKER_SEGMENT = 14, + TRANSCRIBE_ABI_BACKEND_INIT_PARAMS = 15, } transcribe_abi_struct; /* sizeof / alignof of the selected public struct, or 0 for an unknown id. @@ -839,12 +840,12 @@ TRANSCRIBE_API transcribe_status transcribe_init_backends_default(void); * static builds the first device query / model load). A later call with a * different effective mask returns TRANSCRIBE_ERR_BACKEND. Call once, first. */ -#define TRANSCRIBE_BACKEND_MASK_CPU (1u << 0) -#define TRANSCRIBE_BACKEND_MASK_METAL (1u << 1) -#define TRANSCRIBE_BACKEND_MASK_VULKAN (1u << 2) -#define TRANSCRIBE_BACKEND_MASK_CUDA (1u << 3) -#define TRANSCRIBE_BACKEND_MASK_ROCM (1u << 4) -#define TRANSCRIBE_BACKEND_MASK_OTHER (1u << 31) +#define TRANSCRIBE_BACKEND_MASK_CPU 0x00000001u +#define TRANSCRIBE_BACKEND_MASK_METAL 0x00000002u +#define TRANSCRIBE_BACKEND_MASK_VULKAN 0x00000004u +#define TRANSCRIBE_BACKEND_MASK_CUDA 0x00000008u +#define TRANSCRIBE_BACKEND_MASK_ROCM 0x00000010u +#define TRANSCRIBE_BACKEND_MASK_OTHER 0x80000000u #define TRANSCRIBE_BACKEND_MASK_ALL 0xFFFFFFFFu struct transcribe_backend_init_params { diff --git a/src/transcribe.cpp b/src/transcribe.cpp index 28f0bd82c..8f7089c80 100644 --- a/src/transcribe.cpp +++ b/src/transcribe.cpp @@ -236,6 +236,8 @@ extern "C" size_t transcribe_abi_struct_size(transcribe_abi_struct which) { return sizeof(struct transcribe_device_info); case TRANSCRIBE_ABI_SPEAKER_SEGMENT: return sizeof(struct transcribe_speaker_segment); + case TRANSCRIBE_ABI_BACKEND_INIT_PARAMS: + return sizeof(struct transcribe_backend_init_params); } return 0; // unknown id: "cannot verify", never a real size } @@ -272,6 +274,8 @@ extern "C" size_t transcribe_abi_struct_align(transcribe_abi_struct which) { return alignof(struct transcribe_device_info); case TRANSCRIBE_ABI_SPEAKER_SEGMENT: return alignof(struct transcribe_speaker_segment); + case TRANSCRIBE_ABI_BACKEND_INIT_PARAMS: + return alignof(struct transcribe_backend_init_params); } return 0; } @@ -1090,8 +1094,9 @@ static uint32_t env_backend_mask() { if (env == nullptr || env[0] == '\0') { return TRANSCRIBE_BACKEND_MASK_ALL; } - uint32_t m = 0; - const char * tok = env; + uint32_t m = 0; + bool unknown = false; + const char * tok = env; for (const char * p = env;; ++p) { if (*p != '\0' && *p != ',' && *p != ' ' && *p != '\t') { continue; @@ -1104,12 +1109,7 @@ static uint32_t env_backend_mask() { bit = e.bit; } } - if (bit == 0) { - transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_WARN, - "TRANSCRIBE_BACKENDS: ignoring unknown backend '%.*s' " - "(expected cpu, metal, vulkan, cuda, rocm, other, all)", - static_cast(std::min(len, INT_MAX)), tok); - } + unknown = unknown || bit == 0; m |= bit; } if (*p == '\0') { @@ -1117,6 +1117,25 @@ static uint32_t env_backend_mask() { } tok = p + 1; } + // Unknown names are dropped, so a typo narrows (fail-closed). Say + // loudly what that left allowed. + if (unknown) { + char allowed[64] = "all"; + const uint32_t eff = m | TRANSCRIBE_BACKEND_MASK_CPU; + if (eff != TRANSCRIBE_BACKEND_MASK_ALL) { + size_t off = 0; + for (const auto & e : k_tokens) { + if (e.bit != TRANSCRIBE_BACKEND_MASK_ALL && (eff & e.bit) != 0) { + off += static_cast( + std::snprintf(allowed + off, sizeof(allowed) - off, "%s%s", off ? "," : "", e.name)); + } + } + } + transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, + "TRANSCRIBE_BACKENDS='%s': unknown backend name(s) ignored; it now allows: %s " + "(valid: cpu, metal, vulkan, cuda, rocm, other, all)", + env, allowed); + } return m; }(); return mask; @@ -1132,8 +1151,11 @@ static bool backend_reg_filter(const char * name) noexcept { const uint32_t bit = module_mask_bit(name != nullptr ? name : ""); const bool allowed = (effective_backend_mask(host) & bit) != 0; if (!allowed) { - transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_DEBUG, "backend '%s' not registered: excluded by allowed-backend mask", - name != nullptr ? name : "(null)"); + // Callback-only, like the post-scan device summary: no stderr noise. + char msg[160]; + std::snprintf(msg, sizeof(msg), "backend '%s' not registered: excluded by allowed-backend mask", + name != nullptr ? name : "(null)"); + transcribe_log_emit(TRANSCRIBE_LOG_LEVEL_DEBUG, msg); } return allowed; }