From 28fef18043969aa87bdec3615659408d7c790d1c Mon Sep 17 00:00:00 2001 From: Zhenshan Xie Date: Sun, 20 Sep 2026 13:44:15 -0700 Subject: [PATCH] docs(qwen3_8): update NVFP4 checkpoint reference to nvidia/Qwen3.8-27B-NVFP4 RadixArk/Qwen3.8-27B-NVFP4 and nvidia/Qwen3.8-27B-NVFP4 are the same ModelOpt MIXED_PRECISION export; calibrate_qwen3_8_nvfp4() itself never hardcoded either name, since it reads generic HF tensor names from whatever checkpoint directory is passed in. nvidia's checkpoint is now the one tested against and recommended, so update the three comment/ docstring references that still named RadixArk's mirror. Signed-off-by: Zhenshan Xie --- families/qwen3_8/checkpoint_mapper.py | 2 +- families/qwen3_8/quantization.py | 4 ++-- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/families/qwen3_8/checkpoint_mapper.py b/families/qwen3_8/checkpoint_mapper.py index dfe3fcf16b..1c23b92ac5 100644 --- a/families/qwen3_8/checkpoint_mapper.py +++ b/families/qwen3_8/checkpoint_mapper.py @@ -166,7 +166,7 @@ def _apply_block_scales(values: np.ndarray, scale_inv: np.ndarray) -> np.ndarray -# ModelOpt MIXED_PRECISION checkpoints (RadixArk/Qwen3.8-27B-NVFP4) carry two +# ModelOpt MIXED_PRECISION checkpoints (nvidia/Qwen3.8-27B-NVFP4) carry two # schemes side by side, described by quantization_config.config_groups: # # FP8 attention and DeltaNet projections: float8_e4m3 weights with a single diff --git a/families/qwen3_8/quantization.py b/families/qwen3_8/quantization.py index bca9c0c853..fc6ebfc736 100644 --- a/families/qwen3_8/quantization.py +++ b/families/qwen3_8/quantization.py @@ -3,7 +3,7 @@ """Qwen3.8-owned NVFP4 + FP8 TensorRT Q/DQ graph context. -RadixArk/Qwen3.8-27B-NVFP4 is a ModelOpt MIXED_PRECISION export that carries +nvidia/Qwen3.8-27B-NVFP4 is a ModelOpt MIXED_PRECISION export that carries two quantization schemes side by side: NVFP4 MLP projections (gate/up/down) and lm_head: E2M1 values packed two @@ -474,7 +474,7 @@ def calibrate_qwen3_8_nvfp4( if not scales: raise RuntimeError( "Qwen3.8 quantization calibration found no quantized tensors in " - "the checkpoint; is this a RadixArk/Qwen3.8-27B-NVFP4-style " + "the checkpoint; is this an nvidia/Qwen3.8-27B-NVFP4-style " "checkpoint?" )