From 0c45c0a2a29204ca2489d1448f5e99d9e4003699 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Mateusz=20S=C5=82uszniak?= Date: Thu, 24 Sep 2026 12:03:43 +0200 Subject: [PATCH] fix(llm): pin the model load mode instead of inheriting upstream defaults Neither factory was called with a load mode, so both inherited upstream's, and the two defaults are wrong in different ways. The multimodal one defaults to LoadMode::File, which reads the whole .pte into a heap buffer: with gemma4_e2b_mlx_int4.pte that walks phys_footprint past 3.3 GB and jetsam kills the app during load. The text one defaults to MmapUseMlockIgnoreErrors, which leaves footprint alone but wires the model resident: resident_size reads 3135-3282 MB against 355-475 MB for plain Mmap. That is the number DeviceInfo.getUsedMemory() and every other memory API reachable from JS reports, so the new API appeared to use 6x the memory of the legacy one on the same model. Pin Mmap for both, which is what the legacy binding has always passed. --- .../cpp/extensions/llm/llm_runner.cpp | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/packages/react-native-executorch/cpp/extensions/llm/llm_runner.cpp b/packages/react-native-executorch/cpp/extensions/llm/llm_runner.cpp index 396b64c782..8212f3f3b0 100644 --- a/packages/react-native-executorch/cpp/extensions/llm/llm_runner.cpp +++ b/packages/react-native-executorch/cpp/extensions/llm/llm_runner.cpp @@ -11,6 +11,7 @@ #include #include #include +#include #include #include @@ -254,10 +255,16 @@ LLMRunnerHostObject::LLMRunnerHostObject(const std::string &modelPath, throw error::LoadFailed(std::format("LLMRunner: Failed to load runner tokenizer at path: {}", tokenizerPath)); } + // Plain mmap leaves the weights as clean, file-backed pages the kernel + // can evict, which is what the legacy binding has always passed. + constexpr auto kLoadMode = executorch::extension::Module::LoadMode::Mmap; + if (modalities_.empty()) { - runner_ = executorch::extension::llm::create_text_llm_runner(modelPath, std::move(tokenizer)); + runner_ = executorch::extension::llm::create_text_llm_runner( + modelPath, std::move(tokenizer), std::nullopt, -1.0f, "forward", kLoadMode); } else { - runner_ = executorch::extension::llm::create_multimodal_runner(modelPath, std::move(tokenizer)); + runner_ = executorch::extension::llm::create_multimodal_runner( + modelPath, std::move(tokenizer), std::nullopt, kLoadMode); } if (!runner_) {