From a6ce2c28791cf7fbab42db2998518ad90d0c7184 Mon Sep 17 00:00:00 2001 From: zhanghao Date: Thu, 8 Oct 2026 20:11:26 +0800 Subject: [PATCH 1/3] feat(qwen3-8b): first Ascend variant -- Qwen3-8B on one 910B card The catalog has had vendor support in the schema from the start (requires.vendor accepts ascend, and the README maps it to huawei.com/Ascend910), but no variant had ever used it. This is the first non-NVIDIA entry. Qwen3-8B because it fits on a single card, which makes it the cheapest way to prove that a non-NVIDIA node can actually serve: one card, one pod, no tensor-parallel topology to get wrong first. The image is upstream vllm-ascend. The digest is the multi-arch index rather than a platform manifest, so a pull still resolves to arm64 -- which is the one that matters, since Atlas 800T A2 is aarch64 and an amd64-only image would not run there at all. Two NVIDIA-specific values from the other qwen variants are deliberately absent: NVIDIA_DISABLE_REQUIRE is a CUDA-image concern, and PYTORCH_ALLOC_CONF configures the CUDA caching allocator, which torch_npu does not use. No Kubernetes resource strings appear in values: requires.vendor is what decides the extended resource, and that mapping belongs to swiss. extraArgs are derived from the other qwen variants and from what vLLM needs for this model, NOT measured on 910B hardware -- see the PR description. --- models/qwen3-8b/metadata.yaml | 11 ++++ models/qwen3-8b/qwen3-8b-1.0.0.yaml | 83 +++++++++++++++++++++++++++++ 2 files changed, 94 insertions(+) create mode 100644 models/qwen3-8b/metadata.yaml create mode 100644 models/qwen3-8b/qwen3-8b-1.0.0.yaml diff --git a/models/qwen3-8b/metadata.yaml b/models/qwen3-8b/metadata.yaml new file mode 100644 index 0000000..7b36ebb --- /dev/null +++ b/models/qwen3-8b/metadata.yaml @@ -0,0 +1,11 @@ +apiVersion: catalog.swiss/v1 +name: qwen3-8b +displayName: Qwen3-8B +description: Qwen3-8B in bf16 on one Ascend 910B card, served by vLLM's Ascend backend. Small enough for a single card, which is what makes it the model to prove a non-NVIDIA node with. +family: qwen +tags: + - chat + - reasoning + - tool-use +source: + hf: Qwen/Qwen3-8B diff --git a/models/qwen3-8b/qwen3-8b-1.0.0.yaml b/models/qwen3-8b/qwen3-8b-1.0.0.yaml new file mode 100644 index 0000000..758bf5e --- /dev/null +++ b/models/qwen3-8b/qwen3-8b-1.0.0.yaml @@ -0,0 +1,83 @@ +# Serving configs measured, and where marked tuned, by LLM AutoTune. +apiVersion: catalog.swiss/v1 +name: qwen3-8b +version: 1.0.0 +servedName: Qwen3-8B +variants: + - id: vllm-tp1-910b + default: true + description: 'Baseline: one Ascend 910B card, single node. vLLM''s Ascend backend (CANN), not a CUDA build. Not yet tuned -- no AutoTune run on this hardware.' + engine: vllm + chart: + name: vllm + version: ">=0.7.6" + image: + # Upstream vllm-ascend. Multi-arch, and arm64 is the one that matters: + # Atlas 800T A2 is aarch64. The digest is the index, not a platform + # manifest, so a pull still picks the right architecture. + repository: quay.io/ascend/vllm-ascend + tag: v0.23.0.post1 + digest: sha256:ffe9306186781ddb80812eda574134e601ee8499693ea36e4a97173111ddf7e7 + requires: + gpus: 1 + topology: single-node + # Decides the extended resource and the product label; the mapping to + # huawei.com/Ascend910 lives in swiss, not in this file. + vendor: ascend + values: + model: + hostPathType: Directory + contextLength: '' + modelCheck: + enabled: true + requiredGlobs: + - config.json + - '*.safetensors' + extraArgs: + - --enable-prompt-tokens-details + - --reasoning-parser=qwen3 + - --enable-auto-tool-choice + - --tool-call-parser=hermes + - --max-model-len=32768 + - --gpu-memory-utilization=0.9 + env: + # No NVIDIA_DISABLE_REQUIRE and no PYTORCH_ALLOC_CONF here: the first is + # a CUDA-image concern, the second configures the CUDA caching allocator, + # which torch_npu does not use. + - name: VLLM_USE_V1 + value: '1' + volumes: + - name: dot-cache + emptyDir: + sizeLimit: 10Gi + - name: dshm + emptyDir: + medium: Memory + sizeLimit: 16Gi + volumeMounts: + - name: dot-cache + mountPath: /root/.cache + - name: dshm + mountPath: /dev/shm + resources: + requests: + ephemeral-storage: 16Gi + limits: + ephemeral-storage: 16Gi + progressDeadlineSeconds: 7200 + startupProbe: + httpGet: + path: /health + port: http + periodSeconds: 30 + timeoutSeconds: 10 + failureThreshold: 180 + terminationGracePeriodSeconds: 3600 + lifecycle: + preStop: + drainSeconds: 600 + pollIntervalSeconds: 5 + forceShutdown: true + preStopKill: true + hangWatcher: + enabled: true From 0eeadd4c13807a78435b8cb3df173c65c719e777 Mon Sep 17 00:00:00 2001 From: zhanghao Date: Fri, 9 Oct 2026 09:47:28 +0800 Subject: [PATCH 2/3] fix(qwen3-8b): lifecycle uses vllm's keys, not sglang's The values were adapted from qwen3.8-27b-fp8, which is an sglang variant, so they carried lifecycle.forceShutdown -- a key the vllm chart's values.schema.json does not have. Rendering failed outright: at '/lifecycle': additional properties 'forceShutdown' not allowed vllm expresses the same intent as shutdownTimeout, its --shutdown-timeout flag: at 0 vLLM aborts in-flight requests on SIGTERM, above 0 it drains them under that ceiling. Modelled on kimi-k3's vllm-tp8-b300, the only other vllm variant here that configures a lifecycle. terminationGracePeriodSeconds 3600 -> 1200, which is drainSeconds 600 plus the shutdown ceiling plus room. 3600 was inherited from a TP8 variant and is far more than a single-card 8B needs. Rendered against the vllm chart to confirm, rather than trusting the catalog schema: it types values as "any object" and so cannot catch a key the chart rejects. terminationGracePeriodSeconds 1200, preStop present, --shutdown-timeout 120. Putting forceShutdown back reproduces the original failure, so the check discriminates. --- models/qwen3-8b/qwen3-8b-1.0.0.yaml | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/models/qwen3-8b/qwen3-8b-1.0.0.yaml b/models/qwen3-8b/qwen3-8b-1.0.0.yaml index 758bf5e..5c571ea 100644 --- a/models/qwen3-8b/qwen3-8b-1.0.0.yaml +++ b/models/qwen3-8b/qwen3-8b-1.0.0.yaml @@ -72,12 +72,17 @@ variants: periodSeconds: 30 timeoutSeconds: 10 failureThreshold: 180 - terminationGracePeriodSeconds: 3600 + terminationGracePeriodSeconds: 1200 lifecycle: + # vllm's lifecycle keys, not sglang's: there is no forceShutdown here, + # and draining in-flight requests on SIGTERM is what shutdownTimeout + # does -- vLLM aborts them outright when it is 0. preStop: + enabled: true drainSeconds: 600 + endpointSyncSeconds: 10 pollIntervalSeconds: 5 - forceShutdown: true + shutdownTimeout: 120 preStopKill: true hangWatcher: enabled: true From e4508fdfedd66d3bcbd51fa66b2246a28bda6a5d Mon Sep 17 00:00:00 2001 From: zhanghao Date: Fri, 9 Oct 2026 09:59:35 +0800 Subject: [PATCH 3/3] fix(qwen3-8b): keep only the flags that can be justified Four of the six extraArgs did not hold up: --tool-call-parser=hermes guessed from the folklore that Qwen uses the Hermes format. Every other vllm variant here pairs same-family parsers (deepseek_v4/deepseek_v4, kimi_k3/kimi_k3); this one did not, which was the tell. --reasoning-parser=qwen3 copied from qwen3.8-27b-fp8 -- an sglang variant, whose parser names are its own. --enable-auto-tool-choice pointless without a tool parser. --max-model-len=32768 no other variant sets it, and the model's config.json already says 32768. Upstream main does register a "qwen3" parser that serves both roles, but the image here is vllm-ascend v0.23.0.post1, and v0.23's vllm/parser holds only abstract_parser and parser_manager -- the per-model parsers and their registered names came in a later refactor. A name that does not exist in the running version fails at startup, so neither parser is worth asserting until someone has run this image. Tags narrow to chat for the same reason: the model can reason and call tools, but this variant configures no parser for either, so the entry should not advertise them. Rendered against the vllm chart: args are --enable-prompt-tokens-details and --gpu-memory-utilization=0.9 on top of what the chart supplies. --- models/qwen3-8b/metadata.yaml | 4 ++-- models/qwen3-8b/qwen3-8b-1.0.0.yaml | 15 +++++++++++---- 2 files changed, 13 insertions(+), 6 deletions(-) diff --git a/models/qwen3-8b/metadata.yaml b/models/qwen3-8b/metadata.yaml index 7b36ebb..cc5f3cd 100644 --- a/models/qwen3-8b/metadata.yaml +++ b/models/qwen3-8b/metadata.yaml @@ -3,9 +3,9 @@ name: qwen3-8b displayName: Qwen3-8B description: Qwen3-8B in bf16 on one Ascend 910B card, served by vLLM's Ascend backend. Small enough for a single card, which is what makes it the model to prove a non-NVIDIA node with. family: qwen +# chat only: Qwen3-8B can reason and call tools, but this variant configures no +# parser for either, so the entry does not claim capabilities it does not serve. tags: - chat - - reasoning - - tool-use source: hf: Qwen/Qwen3-8B diff --git a/models/qwen3-8b/qwen3-8b-1.0.0.yaml b/models/qwen3-8b/qwen3-8b-1.0.0.yaml index 5c571ea..62d9c31 100644 --- a/models/qwen3-8b/qwen3-8b-1.0.0.yaml +++ b/models/qwen3-8b/qwen3-8b-1.0.0.yaml @@ -34,11 +34,18 @@ variants: - config.json - '*.safetensors' extraArgs: + # Deliberately minimal. Everything here is either measured elsewhere or + # the convention in this catalog; nothing is guessed. + # + # No --reasoning-parser / --tool-call-parser: the parser names cannot be + # confirmed for the vLLM this image carries (v0.23's vllm/parser holds + # only abstract_parser and parser_manager -- the per-model parsers and + # their registered names arrived in a later refactor), and a wrong name + # fails at startup. They belong here once someone has run the image. + # + # No --max-model-len either: no other variant sets it, and Qwen3-8B's + # config.json already declares 32768. - --enable-prompt-tokens-details - - --reasoning-parser=qwen3 - - --enable-auto-tool-choice - - --tool-call-parser=hermes - - --max-model-len=32768 - --gpu-memory-utilization=0.9 env: # No NVIDIA_DISABLE_REQUIRE and no PYTORCH_ALLOC_CONF here: the first is