diff --git a/.cspell-wordlist.txt b/.cspell-wordlist.txt index 882f3cfd22..508fafed04 100644 --- a/.cspell-wordlist.txt +++ b/.cspell-wordlist.txt @@ -6,6 +6,7 @@ executorch RNET libexecutorch libxnnpack +libqnn libvulkan xcframework metallib diff --git a/apps/computer-vision/app/classification/index.tsx b/apps/computer-vision/app/classification/index.tsx index 1dcd140a5a..ee3e12caf5 100644 --- a/apps/computer-vision/app/classification/index.tsx +++ b/apps/computer-vision/app/classification/index.tsx @@ -26,6 +26,11 @@ const MODEL_OPTIONS: ModelOption[] = [ value: models.classification.EFFICIENTNET_V2_S.COREML_FP16, disabled: Platform.OS !== 'ios', }, + { + label: 'EfficientNetV2-S (QNN A16W8)', + value: models.classification.EFFICIENTNET_V2_S.QNN_A16W8, + disabled: Platform.OS !== 'android', + }, ]; function ClassificationContent() { diff --git a/apps/computer-vision/app/segmentation/index.tsx b/apps/computer-vision/app/segmentation/index.tsx index 8222bfa60a..3f85372d61 100644 --- a/apps/computer-vision/app/segmentation/index.tsx +++ b/apps/computer-vision/app/segmentation/index.tsx @@ -1,5 +1,5 @@ import React, { useState, useRef } from 'react'; -import { View, Text, ScrollView } from 'react-native'; +import { View, Text, ScrollView, Platform } from 'react-native'; import { useSafeAreaInsets } from 'react-native-safe-area-context'; import { commonStyles, theme } from '../../theme'; import { @@ -27,26 +27,56 @@ const SEGMENTATION_OPTIONS: ModelOption[] = [ label: 'LRASPP MobileNet V3 (INT8)', value: models.semanticSegmentation.LRASPP_MOBILENET_V3_LARGE.XNNPACK_INT8, }, + { + label: 'LRASPP MobileNet V3 (QNN A16W8)', + value: models.semanticSegmentation.LRASPP_MOBILENET_V3_LARGE.QNN_A16W8, + disabled: Platform.OS !== 'android', + }, { label: 'DeepLab V3 ResNet50 (INT8)', value: models.semanticSegmentation.DEEPLAB_V3_RESNET50.XNNPACK_INT8, }, + { + label: 'DeepLab V3 ResNet50 (QNN A16W8)', + value: models.semanticSegmentation.DEEPLAB_V3_RESNET50.QNN_A16W8, + disabled: Platform.OS !== 'android', + }, { label: 'DeepLab V3 ResNet101 (INT8)', value: models.semanticSegmentation.DEEPLAB_V3_RESNET101.XNNPACK_INT8, }, + { + label: 'DeepLab V3 ResNet101 (QNN A16W8)', + value: models.semanticSegmentation.DEEPLAB_V3_RESNET101.QNN_A16W8, + disabled: Platform.OS !== 'android', + }, { label: 'DeepLab V3 MobileNet V3 (INT8)', value: models.semanticSegmentation.DEEPLAB_V3_MOBILENET_V3_LARGE.XNNPACK_INT8, }, + { + label: 'DeepLab V3 MobileNet V3 (QNN A16W8)', + value: models.semanticSegmentation.DEEPLAB_V3_MOBILENET_V3_LARGE.QNN_A16W8, + disabled: Platform.OS !== 'android', + }, { label: 'FCN ResNet50 (INT8)', value: models.semanticSegmentation.FCN_RESNET50.XNNPACK_INT8, }, + { + label: 'FCN ResNet50 (QNN A16W8)', + value: models.semanticSegmentation.FCN_RESNET50.QNN_A16W8, + disabled: Platform.OS !== 'android', + }, { label: 'FCN ResNet101 (INT8)', value: models.semanticSegmentation.FCN_RESNET101.XNNPACK_INT8, }, + { + label: 'FCN ResNet101 (QNN A16W8)', + value: models.semanticSegmentation.FCN_RESNET101.QNN_A16W8, + disabled: Platform.OS !== 'android', + }, ]; function SegmentationContent() { diff --git a/docs/docs/03-core-and-advanced/08-native-libraries.md b/docs/docs/03-core-and-advanced/08-native-libraries.md index 39d255429e..90bb154ab1 100644 --- a/docs/docs/03-core-and-advanced/08-native-libraries.md +++ b/docs/docs/03-core-and-advanced/08-native-libraries.md @@ -3,15 +3,25 @@ title: Native Libraries slug: /core-and-advanced/native-libraries description: 'How React Native ExecuTorch downloads, ships, and links native binaries on demand.' keywords: - [react native executorch, executorch, native libraries, backends, xnnpack, coreml, mlx, vulkan] + [ + react native executorch, + executorch, + native libraries, + backends, + xnnpack, + coreml, + mlx, + vulkan, + qnn, + ] --- React Native ExecuTorch ships the core runtime, hardware-accelerated backends -(XNNPACK, Core ML, MLX, Vulkan), and native third-party libraries (OpenCV, +(XNNPACK, Core ML, MLX, Vulkan, QNN), and native third-party libraries (OpenCV, phonemis) as **separate downloadable artifacts**. -By default, **everything is downloaded and enabled**, so no configuration is -required to get started. However, because on-device AI backends and vision +By default, **everything except QNN is downloaded and enabled**, so no +configuration is required to get started. However, because on-device AI backends and vision libraries add substantial binary weight, you can tailor exactly what gets pulled into your app. Declaring only the features or backends you use reduces install times, speeds up builds, and significantly shrinks the final app bundle. @@ -42,7 +52,7 @@ Add a `react-native-executorch` block to your `package.json`: | Option | Purpose | Accepted values | | ---------- | ------------------------------------------------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | | `features` | High-level tasks — each automatically expands to the backends and libraries it needs | `"classification"`, `"imageEmbeddings"`, `"instanceSegmentation"`, `"keypointDetection"`, `"llm"`, `"multimodalLLM"`, `"objectDetection"`, `"ocr"`, `"privacyFilter"`, `"segmentAnything"`, `"semanticSegmentation"`, `"speechToText"`, `"styleTransfer"`, `"textEmbeddings"`, `"textToImage"`, `"textToSpeech"`, `"tokenizer"`, `"vad"`, `"verticalOCR"` | -| `backends` | Hardware acceleration backends directly | `"xnnpack"`, `"coreml"`, `"mlx"`, `"vulkan"` | +| `backends` | Hardware acceleration backends directly | `"xnnpack"`, `"coreml"`, `"mlx"`, `"vulkan"`, `"qnn"` | | `libs` | Extra native C++ libraries | `"opencv"`, `"phonemis"` | The three lists are merged, so you can pair high-level `features` with specific `backends` or `libs`. Re-run your package manager install after editing. @@ -71,6 +81,7 @@ Hardware backends provide optimized execution kernels for specific processors an - **[Core ML](https://docs.pytorch.org/executorch/stable/backends/coreml/coreml-overview.html)** — Apple's framework for hardware-accelerated machine learning on Apple Silicon, targeting the Apple Neural Engine (ANE) and GPU. Supported on **iOS only**. - **[MLX](https://github.com/ml-explore/mlx)** — An array framework designed for efficient machine learning on Apple silicon via Metal compute shaders, used primarily for accelerated LLM generation. Supported on **iOS only** (physical device only, no simulator). - **[Vulkan](https://docs.pytorch.org/executorch/stable/backends/vulkan/vulkan-overview.html)** — Cross-platform 3D graphics and compute API, leveraging mobile GPUs on **Android only** for accelerated neural network inference and tensor compute operations. +- **[QNN](https://docs.pytorch.org/executorch/stable/backends-qualcomm.html)** — Qualcomm AI Engine Direct, running models on the Hexagon NPU of Snapdragon 8 Gen 1 and newer (SM8450, SM8475, SM8550, SM8650, SM8750, SM8845, SM8850). **Android arm64 only**, and **opt-in**: it is never enabled by default and no `feature` expands to it, so add `"qnn"` to `backends` explicitly. See [QNN](#qnn) below. ### Third-Party Libraries @@ -103,6 +114,29 @@ Specifying a task under `features` is shorthand: it automatically expands to the | `segmentAnything` | xnnpack, coreml, vulkan | opencv | | `tokenizer` | — | — | +## QNN + +Enabling `qnn` links the ExecuTorch QNN backend and adds Qualcomm's runtime +(`com.qualcomm.qti:qnn-runtime`) from Maven. A QNN model is compiled for one +Hexagon version, so each QNN variant is published once per version and the +registry resolves it to the device's file. On any other device, or when the +requirements below are not met, `DEFAULT` falls back to the next backend. + +The Hexagon DSP loads Qualcomm's skel libraries from disk, so the app has to +extract its native libraries. In `android/app/build.gradle`: + +```groovy +android { + packaging { + jniLibs { + useLegacyPackaging = true + // Not needed for the precompiled models the registry ships. + excludes += ['**/libQnnHtpPrepare.so', '**/libQnnGpu.so', '**/libQnnDsp*.so', '**/libQnnHtpV68*.so'] + } + } +} +``` + ## Binary size Approximate size each backend adds to a release `arm64` build: @@ -113,3 +147,6 @@ Approximate size each backend adds to a release `arm64` build: | `coreml` | — | +0.4 MB | | `mlx` | — | +6.0 MB | | `vulkan` | +10.3 MB | — | +| `qnn` | +92 MB¹ | — | + +¹ Uncompressed, with the excludes shown in [QNN](#qnn); 202 MB without them. diff --git a/packages/react-native-executorch/__tests__/api/modelRegistry.test.ts b/packages/react-native-executorch/__tests__/api/modelRegistry.test.ts index a24ba13fae..1a1f88d0bc 100644 --- a/packages/react-native-executorch/__tests__/api/modelRegistry.test.ts +++ b/packages/react-native-executorch/__tests__/api/modelRegistry.test.ts @@ -105,9 +105,16 @@ const namedVariants = (group: Node): Node[] => const BACKENDS = /^(xnnpack|coreml|mlx|qnn|vulkan)$/; // `spinquant` is a quantization recipe rather than a plain precision, but it is // what the published Llama builds are named after. -const PRECISIONS = /^(fp32|fp16|bf16|int8|int4|8da4w|8da8w|4w|dynamic|spinquant)$/; - -const basename = (url: string) => url.split('/').pop()!.replace('.pte', ''); +const PRECISIONS = /^(fp32|fp16|bf16|int8|int4|8da4w|8da8w|4w|a16w8|dynamic|spinquant)$/; + +// A QNN .pte targets one Hexagon version and carries it as a trailing `_vNN`, +// after the usual modelname_backend_precision. +const basename = (url: string) => + url + .split('/') + .pop()! + .replace('.pte', '') + .replace(/(_qnn_[a-z0-9]+)_v\d+$/, '$1'); /** The backend a `.pte` filename declares, or `undefined` when it declares none. */ const backendOf = (url: string): string | undefined => { diff --git a/packages/react-native-executorch/__tests__/api/modelVariants.test.ts b/packages/react-native-executorch/__tests__/api/modelVariants.test.ts index 8c82b22af3..a6ceb4eb9e 100644 --- a/packages/react-native-executorch/__tests__/api/modelVariants.test.ts +++ b/packages/react-native-executorch/__tests__/api/modelVariants.test.ts @@ -79,11 +79,13 @@ function registryFor(options: { os: 'ios' | 'android'; backends?: string[]; isEmulator?: boolean; + qnnHtpArch?: string; }): Node { fakeJsi.setRegisteredBackends( options.backends ?? ['XnnpackBackend', 'CoreMLBackend', 'MLXBackend', 'VulkanBackend'] ); fakeJsi.setIsEmulator(options.isEmulator ?? false); + fakeJsi.setQnnHtpArch(options.qnnHtpArch); jest.resetModules(); setPlatform(options.os); @@ -326,6 +328,47 @@ describe('variant selection rules', () => { }); }); +describe('QNN variants', () => { + const ANDROID_WITH_QNN = ['XnnpackBackend', 'VulkanBackend', 'QnnBackend']; + const efficientnet = (registry: Node): Node => + (registry.classification as Node).EFFICIENTNET_V2_S as Node; + + it('prefers QNN on Android when the backend is linked and the SoC is known', () => { + const registry = registryFor({ os: 'android', backends: ANDROID_WITH_QNN, qnnHtpArch: 'v75' }); + expect(defaultKeyOf(efficientnet(registry))).toBe('QNN_A16W8'); + }); + + it("resolves a QNN variant to the device's Hexagon version", () => { + const registry = registryFor({ os: 'android', backends: ANDROID_WITH_QNN, qnnHtpArch: 'v75' }); + expect(modelPathsOf(efficientnet(registry).DEFAULT as Node)).toMatch( + /\/qnn\/efficientnet_v2_s_qnn_a16w8_v75\.pte$/ + ); + }); + + it('skips QNN on a device that cannot run it, even with the backend linked', () => { + const registry = registryFor({ os: 'android', backends: ANDROID_WITH_QNN }); + expect(defaultKeyOf(efficientnet(registry))).toBe('XNNPACK_INT8'); + }); + + it('skips QNN when the app did not link the backend', () => { + const registry = registryFor({ os: 'android', qnnHtpArch: 'v81' }); + expect(defaultKeyOf(efficientnet(registry))).toBe('XNNPACK_INT8'); + }); + + it('never picks QNN on iOS', () => { + const registry = registryFor({ + os: 'ios', + backends: ['XnnpackBackend', 'CoreMLBackend', 'QnnBackend'], + qnnHtpArch: 'v81', + }); + expect( + defaultsOf(registry) + .filter(({ key }) => key?.startsWith('QNN')) + .map(({ label, key }) => `${label}: ${key}`) + ).toEqual([]); + }); +}); + describe('feature map', () => { // `models...DEFAULT` only reaches the accelerated export when // the app downloaded that backend, and `features` is the documented way to diff --git a/packages/react-native-executorch/__tests__/support/fakeJsi.ts b/packages/react-native-executorch/__tests__/support/fakeJsi.ts index 1b145c554e..eb8c4cdba0 100644 --- a/packages/react-native-executorch/__tests__/support/fakeJsi.ts +++ b/packages/react-native-executorch/__tests__/support/fakeJsi.ts @@ -327,6 +327,7 @@ let registeredBackends: string[] = ['XnnpackBackend', 'CoreMLBackend']; const jsi = { isEmulator: false, + qnnHtpArch: undefined as string | undefined, createTensor: (shape: number[], dtype: DType) => new FakeTensor(dtype, shape), @@ -434,6 +435,15 @@ export const fakeJsi = { jsi.isEmulator = value; }, + /** + * Sets the Hexagon version the native installer would report for QNN. + * @param value The version (e.g. `'v81'`), or `undefined` for a device that + * cannot run QNN models. + */ + setQnnHtpArch(value: string | undefined): void { + jsi.qnnHtpArch = value; + }, + /** @returns Every `execute` call made so far, in order. */ executions(): readonly { path: string; methodName: string }[] { return executions; @@ -471,6 +481,7 @@ export const fakeJsi = { runnerCalls.length = 0; registeredBackends = ['XnnpackBackend', 'CoreMLBackend']; jsi.isEmulator = false; + jsi.qnnHtpArch = undefined; fakePhonemizer.reset(); tensorTracker.reset(); }, diff --git a/packages/react-native-executorch/android/CMakeLists.txt b/packages/react-native-executorch/android/CMakeLists.txt index 56793ecd17..bbd4749b16 100644 --- a/packages/react-native-executorch/android/CMakeLists.txt +++ b/packages/react-native-executorch/android/CMakeLists.txt @@ -23,6 +23,7 @@ option(RNE_ENABLE_OPENCV "Compile + link OpenCV-dependent sources" ON) option(RNE_ENABLE_PHONEMIS "Compile phonemis (TTS) sources" ON) option(RNE_ENABLE_XNNPACK "Link the XNNPACK backend shared library" ON) option(RNE_ENABLE_VULKAN "Link the Vulkan backend shared library" ON) +option(RNE_ENABLE_QNN "Link the Qualcomm QNN backend shared library (arm64 only)" OFF) find_package(ReactAndroid REQUIRED CONFIG) @@ -169,6 +170,15 @@ if(RNE_ENABLE_VULKAN) IMPORTED_LOCATION "${LIBS_DIR}/executorch/${ANDROID_ABI}/libvulkan_executorch_backend.so") endif() +# ------- QNN backend (optional, arm64 only) ------- +# The Qualcomm runtime it dlopens at load time (libQnnHtp.so and the Hexagon +# skels) comes from the com.qualcomm.qti:qnn-runtime Maven artifact. +if(RNE_ENABLE_QNN AND ANDROID_ABI STREQUAL "arm64-v8a") + add_library(qnn_executorch_backend SHARED IMPORTED) + set_target_properties(qnn_executorch_backend PROPERTIES + IMPORTED_LOCATION "${LIBS_DIR}/executorch/${ANDROID_ABI}/libqnn_executorch_backend.so") +endif() + # ------- OpenCV (optional, static) ------- set(OPENCV_LINK_LIBS "") if(RNE_ENABLE_OPENCV) @@ -224,3 +234,6 @@ endif() if(RNE_ENABLE_VULKAN) target_link_libraries(${CMAKE_PROJECT_NAME} vulkan_executorch_backend) endif() +if(TARGET qnn_executorch_backend) + target_link_libraries(${CMAKE_PROJECT_NAME} qnn_executorch_backend) +endif() diff --git a/packages/react-native-executorch/android/build.gradle.kts b/packages/react-native-executorch/android/build.gradle.kts index f5af7ee6a4..fc21ab1558 100644 --- a/packages/react-native-executorch/android/build.gradle.kts +++ b/packages/react-native-executorch/android/build.gradle.kts @@ -52,6 +52,8 @@ fun rneBuildConfig(): Map<*, *> { val rneConfig = rneBuildConfig() fun rneFlag(key: String): String = if (rneConfig[key] != false) "ON" else "OFF" +// QNN is opt-in, so a config written before it existed (no key) means off. +val enableQnn = rneConfig["enableQnn"] == true /** * The prebuilt ExecuTorch runtime is not in the npm tarball - `package.json` @@ -126,7 +128,8 @@ android { "-DRNE_ENABLE_OPENCV=${rneFlag("enableOpencv")}", "-DRNE_ENABLE_PHONEMIS=${rneFlag("enablePhonemis")}", "-DRNE_ENABLE_XNNPACK=${rneFlag("enableXnnpack")}", - "-DRNE_ENABLE_VULKAN=${rneFlag("enableVulkan")}" + "-DRNE_ENABLE_VULKAN=${rneFlag("enableVulkan")}", + "-DRNE_ENABLE_QNN=${if (enableQnn) "ON" else "OFF"}" ) abiFilters.addAll(reactNativeArchitectures()) @@ -176,10 +179,11 @@ android { // user who provisions third-party/ by hand gets the same treatment. val backendLibs = mapOf( "enableXnnpack" to "libxnnpack_executorch_backend.so", - "enableVulkan" to "libvulkan_executorch_backend.so" + "enableVulkan" to "libvulkan_executorch_backend.so", + "enableQnn" to "libqnn_executorch_backend.so" ) for ((flag, soName) in backendLibs) { - if (rneConfig[flag] == false) { + if (rneConfig[flag] == false || (flag == "enableQnn" && !enableQnn)) { logger.lifecycle("[RnExecutorch] $flag is off; excluding $soName from the APK") jniLibs.excludes.add("**/$soName") } @@ -212,6 +216,13 @@ dependencies { // to third-party/android/libs/executorch.jar. implementation(files("../third-party/android/libs/executorch.jar")) + // Qualcomm's QNN runtime (libQnnHtp, libQnnSystem and the per-arch Hexagon + // stubs/skels) for the QNN backend. Its version must match the QAIRT SDK the + // backend and the .pte files were built with. + if (enableQnn) { + implementation("com.qualcomm.qti:qnn-runtime:2.47.0") + } + // Recommended for modern Kotlin Android development implementation("androidx.core:core-ktx:1.12.0") } diff --git a/packages/react-native-executorch/android/src/main/AndroidManifest.xml b/packages/react-native-executorch/android/src/main/AndroidManifest.xml index a2f47b6057..e37b3b120c 100644 --- a/packages/react-native-executorch/android/src/main/AndroidManifest.xml +++ b/packages/react-native-executorch/android/src/main/AndroidManifest.xml @@ -1,2 +1,8 @@ + + + + diff --git a/packages/react-native-executorch/android/src/main/java/com/swmansion/rnexecutorch/RnExecutorchModule.kt b/packages/react-native-executorch/android/src/main/java/com/swmansion/rnexecutorch/RnExecutorchModule.kt index 12ac5b8d5c..3c29b76c74 100644 --- a/packages/react-native-executorch/android/src/main/java/com/swmansion/rnexecutorch/RnExecutorchModule.kt +++ b/packages/react-native-executorch/android/src/main/java/com/swmansion/rnexecutorch/RnExecutorchModule.kt @@ -1,5 +1,6 @@ package com.swmansion.rnexecutorch +import android.system.Os import com.facebook.react.bridge.JavaScriptContextHolder import com.facebook.react.bridge.ReactApplicationContext @@ -10,10 +11,24 @@ class RnExecutorchModule(reactContext: ReactApplicationContext) : RnExecutorchSp override fun install(): Boolean { val contextHolder: JavaScriptContextHolder = reactApplicationContext.javaScriptContextHolder ?: return false + exposeQnnSkels() nativeInstall(contextHolder.get()) return true } + // The Hexagon DSP loads the QNN skel libraries by path, searching + // ADSP_LIBRARY_PATH, so it has to name the app's native library dir before + // the first QNN model loads. The skels are only on disk when the app extracts + // its native libs (packaging.jniLibs.useLegacyPackaging = true); without them + // the variable stays unset and the JS side does not offer QNN variants. + private fun exposeQnnSkels() { + val libDir = reactApplicationContext.applicationInfo.nativeLibraryDir ?: return + val hasSkel = java.io.File(libDir).list()?.any { it.startsWith("libQnnHtpV") && it.endsWith("Skel.so") } == true + if (!hasSkel) return + val defaults = "/vendor/dsp/cdsp;/vendor/lib/rfsa/adsp;/system/lib/rfsa/adsp;/dsp" + Os.setenv("ADSP_LIBRARY_PATH", "$libDir;$defaults", true) + } + companion object { const val NAME = "RnExecutorch" diff --git a/packages/react-native-executorch/cpp/core/install.cpp b/packages/react-native-executorch/cpp/core/install.cpp index f8ca4c59d5..384bef5352 100644 --- a/packages/react-native-executorch/cpp/core/install.cpp +++ b/packages/react-native-executorch/cpp/core/install.cpp @@ -7,6 +7,7 @@ namespace rnexecutorch::core { void install(facebook::jsi::Runtime &rt, facebook::jsi::Object &module) { utils::install_getExecuTorchRegisteredBackends(rt, module); utils::install_isEmulator(rt, module); + utils::install_qnnHtpArch(rt, module); model::install_loadModel(rt, module); tensor::install_createTensor(rt, module); } diff --git a/packages/react-native-executorch/cpp/core/utils.cpp b/packages/react-native-executorch/cpp/core/utils.cpp index b0a0fd0f10..f79763d0b2 100644 --- a/packages/react-native-executorch/cpp/core/utils.cpp +++ b/packages/react-native-executorch/cpp/core/utils.cpp @@ -1,7 +1,12 @@ #include "utils.h" +#include +#include #include +#include #include +#include +#include #include #include @@ -18,6 +23,62 @@ namespace rnexecutorch::core::utils { namespace jsi = facebook::jsi; namespace { +#ifdef __ANDROID__ +std::string readProp(const char *key) { +#if __ANDROID_API__ >= 26 + const prop_info *pi = __system_property_find(key); + if (pi == nullptr) { + return ""; + } + std::string result; + __system_property_read_callback( + pi, + [](void *cookie, const char * /*name*/, const char *value, uint32_t /*serial*/) { + *static_cast(cookie) = value; + }, + &result); + return result; +#else + char value[PROP_VALUE_MAX] = {0}; + __system_property_get(key, value); + return {value}; +#endif +} +#endif + +// The Hexagon (HTP) version a QNN .pte has to be compiled for, from the SoC. +// A QNN context binary targets one HTP version, so the registry publishes one +// file per version and picks it from this. Only Snapdragon parts ExecuTorch's +// QNN backend knows (backends/qualcomm/serialization/qc_schema.py) are listed. +// Returns nothing off Qualcomm, on an unlisted SoC, or when the skel for that +// version is not where the DSP will look for it (ADSP_LIBRARY_PATH, set by the +// Android module), since a QNN model could not load then. +std::optional qnnHtpArch() { +#ifdef __ANDROID__ + static const std::unordered_map kSocToHtp = { + {"SM8450", 69}, {"SM8475", 69}, {"SM8550", 73}, {"SM8650", 75}, + {"SM8750", 79}, {"SM8845", 81}, {"SM8850", 81}, + }; + const auto it = kSocToHtp.find(readProp("ro.soc.model")); + if (it == kSocToHtp.end()) { + return std::nullopt; + } + const char *adspPath = std::getenv("ADSP_LIBRARY_PATH"); + if (adspPath == nullptr) { + return std::nullopt; + } + const std::string_view paths{adspPath}; + const std::string firstDir{paths.substr(0, paths.find(';'))}; + std::error_code ec; + if (!std::filesystem::exists(std::format("{}/libQnnHtpV{}Skel.so", firstDir, it->second), ec)) { + return std::nullopt; + } + return std::format("v{}", it->second); +#else + return std::nullopt; +#endif +} + // Detects an Android emulator / iOS simulator. On Android no single property // covers every image, so we check three: the build fingerprint (`generic...` // for AOSP images), the hardware name (`goldfish`/`ranchu` are the QEMU @@ -27,27 +88,6 @@ namespace { // is known at compile time. bool isEmulator() { #ifdef __ANDROID__ - auto readProp = [](const char *key) -> std::string { -#if __ANDROID_API__ >= 26 - const prop_info *pi = __system_property_find(key); - if (pi == nullptr) { - return ""; - } - std::string result; - __system_property_read_callback( - pi, - [](void *cookie, const char * /*name*/, const char *value, uint32_t /*serial*/) { - *static_cast(cookie) = value; - }, - &result); - return result; -#else - char value[PROP_VALUE_MAX] = {0}; - __system_property_get(key, value); - return {value}; -#endif - }; - const auto startsWith = [](const std::string &value, const char *prefix) { return value.rfind(prefix, 0) == 0; }; @@ -105,4 +145,10 @@ void install_getExecuTorchRegisteredBackends(jsi::Runtime &rt, jsi::Object &modu void install_isEmulator(jsi::Runtime &rt, jsi::Object &module) { module.setProperty(rt, "isEmulator", jsi::Value(isEmulator())); } + +void install_qnnHtpArch(jsi::Runtime &rt, jsi::Object &module) { + const auto arch = qnnHtpArch(); + module.setProperty(rt, "qnnHtpArch", + arch ? jsi::Value(jsi::String::createFromUtf8(rt, *arch)) : jsi::Value::undefined()); +} } // namespace rnexecutorch::core::utils diff --git a/packages/react-native-executorch/cpp/core/utils.h b/packages/react-native-executorch/cpp/core/utils.h index ae352c7094..e45945fdf6 100644 --- a/packages/react-native-executorch/cpp/core/utils.h +++ b/packages/react-native-executorch/cpp/core/utils.h @@ -21,4 +21,14 @@ void install_getExecuTorchRegisteredBackends(facebook::jsi::Runtime &rt, faceboo * @param module The `__rnexecutorch_jsi__` module object to install onto. */ void install_isEmulator(facebook::jsi::Runtime &rt, facebook::jsi::Object &module); + +/** + * Installs `qnnHtpArch`, the Hexagon version (`"v69"` … `"v81"`) QNN models + * must be compiled for on this device, or `undefined` when the device cannot + * run them: not a known Snapdragon, or the matching skel is not reachable. + * + * @param rt The active JavaScript runtime. + * @param module The `__rnexecutorch_jsi__` module object to install onto. + */ +void install_qnnHtpArch(facebook::jsi::Runtime &rt, facebook::jsi::Object &module); } // namespace rnexecutorch::core::utils diff --git a/packages/react-native-executorch/package.json b/packages/react-native-executorch/package.json index 87a0fe7cd6..6b4ef4096a 100644 --- a/packages/react-native-executorch/package.json +++ b/packages/react-native-executorch/package.json @@ -1,7 +1,7 @@ { "name": "react-native-executorch", "version": "0.11.0", - "nativeLibsVersion": "0.10.4", + "nativeLibsVersion": "0.10.5", "description": "An easy way to run AI models in React Native with ExecuTorch", "main": "./lib/module/index.js", "module": "./lib/module/index.js", diff --git a/packages/react-native-executorch/scripts/download-libs.js b/packages/react-native-executorch/scripts/download-libs.js index bd76822197..8e48e430cb 100644 --- a/packages/react-native-executorch/scripts/download-libs.js +++ b/packages/react-native-executorch/scripts/download-libs.js @@ -23,6 +23,7 @@ * xnnpack-ios.tar.gz -- XnnpackBackend.xcframework (iOS) * vulkan-android-arm64-v8a.tar.gz -- libvulkan_executorch_backend.so (Android only) * vulkan-android-x86_64.tar.gz + * qnn-android-arm64-v8a.tar.gz -- libqnn_executorch_backend.so (Android arm64 only) * coreml-ios.tar.gz -- CoreMLBackend.xcframework (iOS only) * mlx-ios.tar.gz -- MLXBackend.xcframework + mlx.metallib (iOS only) * @@ -31,16 +32,17 @@ * * User configuration (in the app's package.json) — three optional arrays, all merged into a single set: * "react-native-executorch": { - * "backends": ["xnnpack", "coreml", "mlx", "vulkan"], + * "backends": ["xnnpack", "coreml", "mlx", "vulkan", "qnn"], * "libs": ["opencv", "phonemis"], * "features": ["llm", "textToSpeech", "objectDetection"] * } * * `features` is sugar — each one expands to a set of backends + libs via FEATURE_MAP below. - * If no `react-native-executorch` block is present, every backend and lib defaults to ON. + * If no `react-native-executorch` block is present, every backend and lib defaults to ON, + * except qnn, which is opt-in only (see below). * * Recognized values: - * backends: xnnpack, coreml (iOS), mlx (iOS), vulkan (Android) + * backends: xnnpack, coreml (iOS), mlx (iOS), vulkan (Android), qnn (Android arm64) * libs: opencv, phonemis * features: llm, multimodalLLM, speechToText, textToSpeech, vad, privacyFilter, * textEmbeddings, imageEmbeddings, @@ -56,6 +58,9 @@ * coreml iOS only — toggles CoreMLBackend.xcframework. * mlx iOS only — toggles MLXBackend.xcframework + mlx.metallib resource. * vulkan Android only — toggles libvulkan_executorch_backend.so. + * qnn Android arm64 only — toggles libqnn_executorch_backend.so and the Qualcomm + * runtime (Maven com.qualcomm.qti:qnn-runtime). Opt-in: the runtime adds tens of + * MB to the APK, so an app without a config block does not get it. * * Environment variables: * RNET_SKIP_DOWNLOAD=1 -- skip download entirely (for CI with pre-cached libs) @@ -110,7 +115,11 @@ const CACHE_DIR = process.env.RNET_LIBS_CACHE_DIR || DEFAULT_CACHE_DIR; // ---- User config ----------------------------------------------------------- -const ALL_BACKENDS = ['xnnpack', 'coreml', 'mlx', 'vulkan']; +const ALL_BACKENDS = ['xnnpack', 'coreml', 'mlx', 'vulkan', 'qnn']; +// What an app without a config block gets. qnn stays out: it pulls the +// Qualcomm runtime from Maven, which is far larger than any other backend and +// only pays off on Snapdragon devices. +const DEFAULT_BACKENDS = ALL_BACKENDS.filter((b) => b !== 'qnn'); const ALL_LIBS = ['opencv', 'phonemis']; // features -> { backends, libs } @@ -229,7 +238,11 @@ function findUserConfig() { } function readUserConfig() { - const allOn = () => ({ backends: [...ALL_BACKENDS], libs: [...ALL_LIBS], opencvPod: undefined }); + const allOn = () => ({ + backends: [...DEFAULT_BACKENDS], + libs: [...ALL_LIBS], + opencvPod: undefined, + }); const { config: rneConfig, manifest } = findUserConfig(); @@ -291,6 +304,7 @@ function writeBuildConfig({ backends, libs, opencvPod }) { enableCoreml: backends.includes('coreml'), enableMlx: backends.includes('mlx'), enableVulkan: backends.includes('vulkan'), + enableQnn: backends.includes('qnn'), }; // Which pod provides opencv2 on iOS. Only set when the app asks for a // specific one; otherwise the podspec picks, preferring an OpenCV the app @@ -319,6 +333,11 @@ function warnAboutPlatformAsymmetry({ backends }, targets) { '[react-native-executorch] mlx is enabled but the build targets only Android; the MLX backend is iOS-only and the flag has no effect here.' ); } + if (hasIos && !hasAndroid && backends.includes('qnn')) { + console.warn( + '[react-native-executorch] qnn is enabled but the build targets only iOS; the QNN backend is Android-only and the flag has no effect here.' + ); + } if (hasIos && !hasAndroid && backends.includes('vulkan')) { console.warn( '[react-native-executorch] vulkan is enabled but the build targets only iOS; the Vulkan backend is Android-only and the flag has no effect here.' @@ -396,6 +415,11 @@ function getArtifacts(targets, { backends, libs }) { if (backends.includes('vulkan') && target.startsWith('android')) { artifacts.push(makeArtifact(`vulkan-${target}`, destDir)); } + + // QNN targets Snapdragon's Hexagon NPU, so it is Android arm64 only + if (backends.includes('qnn') && target === 'android-arm64-v8a') { + artifacts.push(makeArtifact(`qnn-${target}`, destDir)); + } } return artifacts; @@ -478,6 +502,7 @@ const BACKEND_FILES = { android: { xnnpack: ['executorch/*/libxnnpack_executorch_backend.so'], vulkan: ['executorch/*/libvulkan_executorch_backend.so'], + qnn: ['executorch/*/libqnn_executorch_backend.so'], }, ios: { xnnpack: ['XnnpackBackend.xcframework'], @@ -562,7 +587,7 @@ async function main() { `[react-native-executorch] Backends: [${config.backends.join(', ') || '—'}]; Libs: [${config.libs.join(', ') || '—'}]` ); console.log( - `[react-native-executorch] Build flags: opencv=${buildConfig.enableOpencv}, phonemis=${buildConfig.enablePhonemis}, xnnpack=${buildConfig.enableXnnpack}, coreml=${buildConfig.enableCoreml}, mlx=${buildConfig.enableMlx}, vulkan=${buildConfig.enableVulkan}` + `[react-native-executorch] Build flags: opencv=${buildConfig.enableOpencv}, phonemis=${buildConfig.enablePhonemis}, xnnpack=${buildConfig.enableXnnpack}, coreml=${buildConfig.enableCoreml}, mlx=${buildConfig.enableMlx}, vulkan=${buildConfig.enableVulkan}, qnn=${buildConfig.enableQnn}` ); const targets = detectTargets(); diff --git a/packages/react-native-executorch/scripts/package-release-artifacts.sh b/packages/react-native-executorch/scripts/package-release-artifacts.sh index 0ebee9dea3..99b45b0f4d 100755 --- a/packages/react-native-executorch/scripts/package-release-artifacts.sh +++ b/packages/react-native-executorch/scripts/package-release-artifacts.sh @@ -16,6 +16,7 @@ # xnnpack-android-x86_64.tar.gz + .sha256 # vulkan-android-arm64-v8a.tar.gz + .sha256 # vulkan-android-x86_64.tar.gz + .sha256 +# qnn-android-arm64-v8a.tar.gz + .sha256 (arm64 only; Qualcomm runtime comes from Maven) # core-ios.tar.gz + .sha256 (ExecutorchLib.xcframework + libthreadpool_*.a) # xnnpack-ios.tar.gz + .sha256 # coreml-ios.tar.gz + .sha256 @@ -218,6 +219,10 @@ package_file "vulkan-android-arm64-v8a" \ package_file "vulkan-android-x86_64" \ "executorch/x86_64" "$ANDROID_LIBS/executorch/x86_64/libvulkan_executorch_backend.so" +# QNN runs on the Snapdragon Hexagon NPU, so there is no x86_64 build. +package_file "qnn-android-arm64-v8a" \ + "executorch/arm64-v8a" "$ANDROID_LIBS/executorch/arm64-v8a/libqnn_executorch_backend.so" + # ---- iOS -------------------------------------------------------------------- # Note: OpenCV for iOS is provided by CocoaPods (opencv-rne dependency). # No opencv-ios tarball is needed. diff --git a/packages/react-native-executorch/src/models.ts b/packages/react-native-executorch/src/models.ts index 1ab72f13a6..903a6770df 100644 --- a/packages/react-native-executorch/src/models.ts +++ b/packages/react-native-executorch/src/models.ts @@ -68,7 +68,7 @@ import { // that order pins one per platform — see the second argument of `variants`. /** Every backend the registry publishes for, spelled as the variant keys spell it. */ -const ALL_BACKENDS = ['xnnpack', 'coreml', 'mlx', 'vulkan'] as const; +const ALL_BACKENDS = ['xnnpack', 'coreml', 'mlx', 'vulkan', 'qnn'] as const; /** The backend prefix a variant key starts with. */ type BackendTag = (typeof ALL_BACKENDS)[number]; @@ -82,15 +82,25 @@ const PLATFORM: TargetPlatform = Platform.OS === 'ios' ? 'ios' : 'android'; // MLX or Vulkan once it has been shown to run better there, and XNNPACK is the // one backend every model exports to. Core ML sits above MLX only to make the // order deterministic; every group publishing both pins its winner explicitly. +// QNN leads on Android: it runs on the Hexagon NPU, and a model only gets a +// QNN export once it beats the GPU there. // // The iOS simulator links the Core ML backend but cannot run it: no Neural // Engine, and MPSGraph refuses the compiled models. MLX only ever ships a // device slice, so it has nothing to run there either. const BACKEND_ORDER: Record = { ios: rnexecutorchJsi.isEmulator === true ? ['xnnpack'] : ['coreml', 'mlx', 'xnnpack'], - android: ['vulkan', 'xnnpack'], + android: ['qnn', 'vulkan', 'xnnpack'], }; +/** + * The Hexagon version (`v69` … `v81`) this device's QNN models are compiled + * for, or `undefined` when it cannot run them. A QNN `.pte` targets a single + * Hexagon version, so QNN variants are published once per version and resolve + * to this device's file. + */ +const QNN_HTP_ARCH: string | undefined = rnexecutorchJsi.qnnHtpArch; + /** * The backends this platform may default to, best first. * @returns This platform's order, less every backend the binary was not linked @@ -107,7 +117,13 @@ function getCandidateBackends(): readonly BackendTag[] { if (registered.length === 0) return BACKEND_ORDER[PLATFORM]; const names = registered.map((name) => name.toLowerCase()); - return BACKEND_ORDER[PLATFORM].filter((tag) => names.some((name) => name.startsWith(tag))); + return BACKEND_ORDER[PLATFORM].filter( + (tag) => + names.some((name) => name.startsWith(tag)) && + // A linked QNN backend is not enough: the SoC also has to be one the + // published files target, with its Hexagon skel reachable. + (tag !== 'qnn' || QNN_HTP_ARCH !== undefined) + ); } const CANDIDATE_BACKENDS = getCandidateBackends(); @@ -173,10 +189,11 @@ function family>( } const BASE_URL = 'https://huggingface.co/software-mansion/react-native-executorch'; -// Models resolve through VERSION_TAG, the latest published stable tag, unless -// they have been re-exported for the next release, in which case their URL moves -// over to NEXT_VERSION_TAG. Both tags are pinned snapshots, so a model only -// changes when its URL is moved here. +// Most models still resolve through VERSION_TAG, the latest published stable +// tag. A model re-exported for 0.11 moves the URLs it re-exported over to +// NEXT_VERSION_TAG. The two coexist because a tag carries only the files cut +// under it, so a backend first published in 0.11 sits on the newer tag while +// the same model's older exports stay on the older one. const VERSION_TAG = 'resolve/v0.10.0'; const NEXT_VERSION_TAG = 'resolve/v0.11.0'; @@ -201,6 +218,12 @@ const EFFICIENTNET_V2_S_COREML_FP16: ClassifierModel = { modelPath: `${BASE_URL}-efficientnet-v2-s/${VERSION_TAG}/coreml/efficientnet_v2_s_coreml_fp16.pte`, modelOpts: EFFICIENTNET_V2_S_OPTS, }; +// Resolves to this device's Hexagon version; off Snapdragon the v81 file stands +// in, and DEFAULT never picks it there. +const EFFICIENTNET_V2_S_QNN_A16W8: ClassifierModel = { + modelPath: `${BASE_URL}-efficientnet-v2-s/${NEXT_VERSION_TAG}/qnn/efficientnet_v2_s_qnn_a16w8_${QNN_HTP_ARCH ?? 'v81'}.pte`, + modelOpts: EFFICIENTNET_V2_S_OPTS, +}; // ============================================================================= // Style Transfer @@ -323,6 +346,12 @@ const LRASPP_MOBILENET_V3_LARGE_COREML_FP16: SemanticSegmenterModel = { + modelPath: `${BASE_URL}-lraspp/${NEXT_VERSION_TAG}/qnn/lraspp_mobilenet_v3_large_qnn_a16w8_${QNN_HTP_ARCH ?? 'v81'}.pte`, + modelOpts: LRASPP_MOBILENET_V3_LARGE_OPTS, +}; const DEEPLAB_V3_OPTS = { labels: PASCAL_VOC_LABELS, @@ -343,6 +372,16 @@ const DEEPLAB_V3_RESNET50_COREML_FP16: SemanticSegmenterModel = modelPath: `${BASE_URL}-deeplab-v3/${NEXT_VERSION_TAG}/coreml/deeplab_v3_resnet50_coreml_fp16.pte`, modelOpts: DEEPLAB_V3_OPTS, }; +// Same Hexagon-version resolution and index-map output as the MobileNetV3 QNN +// variant below. +const DEEPLAB_V3_RESNET50_QNN_A16W8: SemanticSegmenterModel = { + modelPath: `${BASE_URL}-deeplab-v3/${NEXT_VERSION_TAG}/qnn/deeplab_v3_resnet50_qnn_a16w8_${QNN_HTP_ARCH ?? 'v81'}.pte`, + modelOpts: DEEPLAB_V3_OPTS, +}; +const DEEPLAB_V3_RESNET101_QNN_A16W8: SemanticSegmenterModel = { + modelPath: `${BASE_URL}-deeplab-v3/${NEXT_VERSION_TAG}/qnn/deeplab_v3_resnet101_qnn_a16w8_${QNN_HTP_ARCH ?? 'v81'}.pte`, + modelOpts: DEEPLAB_V3_OPTS, +}; const DEEPLAB_V3_RESNET101_XNNPACK_FP32: SemanticSegmenterModel = { modelPath: `${BASE_URL}-deeplab-v3/${VERSION_TAG}/xnnpack/deeplab_v3_resnet101_xnnpack_fp32.pte`, modelOpts: DEEPLAB_V3_OPTS, @@ -367,6 +406,13 @@ const DEEPLAB_V3_MOBILENET_V3_LARGE_COREML_FP16: SemanticSegmenterModel = { + modelPath: `${BASE_URL}-deeplab-v3/${NEXT_VERSION_TAG}/qnn/deeplab_v3_mobilenet_v3_large_qnn_a16w8_${QNN_HTP_ARCH ?? 'v81'}.pte`, + modelOpts: DEEPLAB_V3_OPTS, +}; const FCN_OPTS = { labels: PASCAL_VOC_LABELS, @@ -399,6 +445,16 @@ const FCN_RESNET101_COREML_FP16: SemanticSegmenterModel = { modelPath: `${BASE_URL}-fcn/${NEXT_VERSION_TAG}/coreml/fcn_resnet101_coreml_fp16.pte`, modelOpts: FCN_OPTS, }; +// Same Hexagon-version resolution and index-map output as the DeepLabV3 QNN +// variants. +const FCN_RESNET50_QNN_A16W8: SemanticSegmenterModel = { + modelPath: `${BASE_URL}-fcn/${NEXT_VERSION_TAG}/qnn/fcn_resnet50_qnn_a16w8_${QNN_HTP_ARCH ?? 'v81'}.pte`, + modelOpts: FCN_OPTS, +}; +const FCN_RESNET101_QNN_A16W8: SemanticSegmenterModel = { + modelPath: `${BASE_URL}-fcn/${NEXT_VERSION_TAG}/qnn/fcn_resnet101_qnn_a16w8_${QNN_HTP_ARCH ?? 'v81'}.pte`, + modelOpts: FCN_OPTS, +}; // ============================================================================= // Object Detection @@ -2083,6 +2139,7 @@ export const models = { XNNPACK_INT8: EFFICIENTNET_V2_S_XNNPACK_INT8, XNNPACK_FP32: EFFICIENTNET_V2_S_XNNPACK_FP32, COREML_FP16: EFFICIENTNET_V2_S_COREML_FP16, + QNN_A16W8: EFFICIENTNET_V2_S_QNN_A16W8, }), }, @@ -2164,6 +2221,7 @@ export const models = { XNNPACK_INT8: LRASPP_MOBILENET_V3_LARGE_XNNPACK_INT8, XNNPACK_FP32: LRASPP_MOBILENET_V3_LARGE_XNNPACK_FP32, COREML_FP16: LRASPP_MOBILENET_V3_LARGE_COREML_FP16, + QNN_A16W8: LRASPP_MOBILENET_V3_LARGE_QNN_A16W8, }), /** * DeepLabV3 semantic segmentation model with ResNet-50 backbone (21 @@ -2174,6 +2232,7 @@ export const models = { XNNPACK_INT8: DEEPLAB_V3_RESNET50_XNNPACK_INT8, XNNPACK_FP32: DEEPLAB_V3_RESNET50_XNNPACK_FP32, COREML_FP16: DEEPLAB_V3_RESNET50_COREML_FP16, + QNN_A16W8: DEEPLAB_V3_RESNET50_QNN_A16W8, }), /** * DeepLabV3 semantic segmentation model with ResNet-101 backbone (21 @@ -2184,6 +2243,7 @@ export const models = { XNNPACK_INT8: DEEPLAB_V3_RESNET101_XNNPACK_INT8, XNNPACK_FP32: DEEPLAB_V3_RESNET101_XNNPACK_FP32, COREML_FP16: DEEPLAB_V3_RESNET101_COREML_FP16, + QNN_A16W8: DEEPLAB_V3_RESNET101_QNN_A16W8, }), /** * DeepLabV3 semantic segmentation model with MobileNetV3-Large backbone (21 @@ -2194,6 +2254,7 @@ export const models = { XNNPACK_INT8: DEEPLAB_V3_MOBILENET_V3_LARGE_XNNPACK_INT8, XNNPACK_FP32: DEEPLAB_V3_MOBILENET_V3_LARGE_XNNPACK_FP32, COREML_FP16: DEEPLAB_V3_MOBILENET_V3_LARGE_COREML_FP16, + QNN_A16W8: DEEPLAB_V3_MOBILENET_V3_LARGE_QNN_A16W8, }), /** * Fully Convolutional Network (FCN) semantic segmentation model with @@ -2203,6 +2264,7 @@ export const models = { XNNPACK_INT8: FCN_RESNET50_XNNPACK_INT8, XNNPACK_FP32: FCN_RESNET50_XNNPACK_FP32, COREML_FP16: FCN_RESNET50_COREML_FP16, + QNN_A16W8: FCN_RESNET50_QNN_A16W8, }), /** * Fully Convolutional Network (FCN) semantic segmentation model with @@ -2212,6 +2274,7 @@ export const models = { XNNPACK_INT8: FCN_RESNET101_XNNPACK_INT8, XNNPACK_FP32: FCN_RESNET101_XNNPACK_FP32, COREML_FP16: FCN_RESNET101_COREML_FP16, + QNN_A16W8: FCN_RESNET101_QNN_A16W8, }), },