diff --git a/.DS_Store b/.DS_Store deleted file mode 100644 index 82ca968..0000000 Binary files a/.DS_Store and /dev/null differ diff --git a/.github/dependabot.yml b/.github/dependabot.yml new file mode 100644 index 0000000..c0252bc --- /dev/null +++ b/.github/dependabot.yml @@ -0,0 +1,21 @@ +version: 2 +updates: + - package-ecosystem: gomod + directory: / + schedule: + interval: weekly + groups: + k8s: + patterns: ["k8s.io/*", "sigs.k8s.io/*"] + commit-message: + prefix: "chore(deps)" + + - package-ecosystem: github-actions + directory: / + schedule: + interval: weekly + commit-message: + prefix: "ci" + +# No docker entry: the Dockerfiles take their base images from build args +# (FROM ${GO_BASE}), which Dependabot does not follow. diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..b89b7cc --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,79 @@ +name: ci + +# Tests and repository checks on every pull request and push to main. +# Building and publishing the images is release.yml. +on: + push: + branches: + - main + pull_request: + +permissions: + contents: read + +concurrency: + group: ci-${{ github.ref }} + cancel-in-progress: ${{ github.event_name == 'pull_request' }} + +jobs: + go: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - uses: actions/setup-go@b7ad1dad31e06c5925ef5d2fc7ad053ef454303e # v7.0.0 + with: + go-version-file: go.mod + - name: gofmt + run: | + unformatted=$(gofmt -l .) + if [ -n "$unformatted" ]; then + echo "::error::not gofmt-formatted:"; echo "$unformatted"; exit 1 + fi + - run: go vet ./... + # Regenerates the CRD, RBAC and deepcopy code, then vets and tests. + - run: make test + - name: Generated files are committed + run: | + if ! git diff --exit-code; then + echo "::error::make manifests generate changed files; run them and commit the result." + exit 1 + fi + + helm: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + # helm comes preinstalled on the GitHub-hosted Ubuntu runners. + - run: helm lint --strict deploy/helm/autoconfig + - run: helm template autoconfig deploy/helm/autoconfig > /dev/null + + docker: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - run: docker build -t autoconfig:ci . + + gate: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - name: LICENSE is the Apache-2.0 text, unmodified + run: | + if ! grep -q "Apache License" LICENSE || ! grep -q "Version 2.0, January 2004" LICENSE; then + echo "::error file=LICENSE::LICENSE is not the Apache-2.0 text." + exit 1 + fi + - name: No credentials committed + run: | + if git ls-files | grep -InE '(^|/)\.env$|\.kubeconfig$|\.pem$|(^|/)(id_rsa|id_ed25519)$'; then + echo "::error::a credential file is tracked" + exit 1 + fi diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 2930c15..f9d4df7 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -1,11 +1,9 @@ # Public release: a version tag builds the images and pushes them to Docker Hub. # -# The internal image is built separately by .gitlab-ci.yml and goes to harbor. -# Both build the same Dockerfiles; only the base-image build args and the -# destination registry differ. Every base here already defaults to a public -# image, so this side passes no overrides. +# Every base image and the Go module proxy in the Dockerfiles default to public +# sources, so this workflow passes no build-arg overrides. # -# One tag produces three images, the way the internal pipeline does. +# One tag produces three images: the controller and its two sidecars. # # Credentials come from the organisation secrets DOCKERHUB_USERNAME and # DOCKERHUB_TOKEN. diff --git a/.gitignore b/.gitignore index 56287c3..62464c3 100644 --- a/.gitignore +++ b/.gitignore @@ -7,3 +7,6 @@ autoconfig.exe # 排查时的抓包/落盘,禁止入库(含真实请求体与内网地址) /debug/ + +# macOS Finder metadata +.DS_Store diff --git a/.gitlab-ci.yml b/.gitlab-ci.yml deleted file mode 100644 index 0a40b39..0000000 --- a/.gitlab-ci.yml +++ /dev/null @@ -1,82 +0,0 @@ -# 打 tag 自动 build autoconfig 镜像并 push 到 harbor。 -# -# 参考同组 llm-openresty/chat-engine 惯例:org 共享 `public-buildx` runner —— docker daemon 已内置 -# harbor 鉴权,CI 里不需要 docker login、也不需要配 HARBOR_ 变量,直接 docker build + docker push。 -# -# 镜像多阶段 build(照 llm-openresty/Dockerfile.bodylog):builder = golang:1.23.3-alpine3.20,依赖走国内 goproxy -# (mirrors.tencent.com/go —— public-buildx runner 实测可达,见 llm-openresty CI #416384),不再 vendor(go.sum+GOSUMDB=off 可复现)。 -# -# 触发:仅打 git tag。产物 harbor.4pd.io/hardcore-tech/autoconfig: (+ :latest;预发布 tag 不推 latest)。 -# 生效前提:本文件要在【被打 tag 的 commit】上(先合进 main),再打 tag。 - -stages: [build] - -variables: - IMAGE: harbor.4pd.io/hardcore-tech/autoconfig # controller - RELOAD_IMAGE: harbor.4pd.io/hardcore-tech/autoconfig-reload # reload sidecar(独立小镜像) - HAGATE_IMAGE: harbor.4pd.io/hardcore-tech/autoconfig-hagate # master-standby leader 选举门控 sidecar - - # Dockerfile 里这三个 ARG 的【默认值是公网】(Docker Hub / proxy.golang.org), - # 这样仓库开源后外部用户 `docker build .` 开箱即用。内网 CI 在这里覆盖回 harbor 缓存 - # 与国内 goproxy —— 默认值服务于"什么都不知道的人",内网本来就有配置文件,加参数零成本。 - BUILD_ARGS: >- - --build-arg GO_BASE=harbor.4pd.io/library/golang:1.23.3-alpine3.20 - --build-arg RUNTIME_BASE=harbor.4pd.io/hardcore-tech/python:3.12-alpine - --build-arg GOPROXY=https://mirrors.tencent.com/go/,direct - -# 三个镜像 job 的公共部分。PRERELEASE:tag 里带 "-"(SemVer 预发布,如 0.3.34-slo-rc1)时非空, -# 各 job 据此**跳过 :latest** —— RC 是拿去测试集群验证的,不该变成「没指定版本时默认拿到的那个」。 -# 正式版(0.3.34)行为不变,照旧同时推 :latest。 -.tag-build: - stage: build - tags: [public-buildx] # org 共享构建 runner(在线到 harbor;docker 已 login harbor) - rules: - - if: '$CI_COMMIT_TAG' # 只在打 tag 时构建;其它 push 不触发 - before_script: - - 'case "$CI_COMMIT_TAG" in *-*) PRERELEASE=1 ;; *) PRERELEASE= ;; esac' - - 'echo "tag=$CI_COMMIT_TAG prerelease=${PRERELEASE:-0}"' - -build:image: - extends: .tag-build - script: - - docker build $BUILD_ARGS -t "$IMAGE:$CI_COMMIT_TAG" . - - docker push "$IMAGE:$CI_COMMIT_TAG" - # 预发布跳过 :latest(见 .tag-build 的 PRERELEASE) - - if [ -z "$PRERELEASE" ]; then docker tag "$IMAGE:$CI_COMMIT_TAG" "$IMAGE:latest" && docker push "$IMAGE:latest"; fi - -build:reload: - extends: .tag-build - script: - - docker build $BUILD_ARGS -f Dockerfile.reload -t "$RELOAD_IMAGE:$CI_COMMIT_TAG" . - - docker push "$RELOAD_IMAGE:$CI_COMMIT_TAG" - # 预发布跳过 :latest(见 .tag-build 的 PRERELEASE) - - if [ -z "$PRERELEASE" ]; then docker tag "$RELOAD_IMAGE:$CI_COMMIT_TAG" "$RELOAD_IMAGE:latest" && docker push "$RELOAD_IMAGE:latest"; fi - -build:hagate: - extends: .tag-build - script: - - docker build $BUILD_ARGS -f Dockerfile.hagate -t "$HAGATE_IMAGE:$CI_COMMIT_TAG" . - - docker push "$HAGATE_IMAGE:$CI_COMMIT_TAG" - # 预发布跳过 :latest(见 .tag-build 的 PRERELEASE) - - if [ -z "$PRERELEASE" ]; then docker tag "$HAGATE_IMAGE:$CI_COMMIT_TAG" "$HAGATE_IMAGE:latest" && docker push "$HAGATE_IMAGE:latest"; fi - -# helm chart 打包 + push,用 public-buildx runner 内置的 push-chart(org 标准,参考 llm-monitor CI 实测)。 -# push-chart 自处理 harbor ChartMuseum 鉴权(helm cm-push 到 runner 内 helm repo 别名 harbor-chart-repo, -# 即 https://harbor.4pd.io/chartrepo/hardcore-tech);不用 docker login、不用 helm 镜像、不用新凭证。 -# 版本:autoconfig 的 chart version=appVersion=git tag(三者同线),把 Chart.yaml 两行 sed 成 $CI_COMMIT_TAG -# → values.yaml image.tag="" 回落 appVersion → controller 镜像 tag 与 chart 版本永远一致。 -# (reload/hagate 侧车镜像在 values 里另钉版本,不随本 tag 走;需同步升时手动改 values。) -# 消费:helm repo add harbor-chart-repo https://harbor.4pd.io/chartrepo/hardcore-tech && helm upgrade ... autoconfig --version -build:chart: - stage: build - tags: [public-buildx] - rules: - - if: '$CI_COMMIT_TAG' - variables: - MODULE_NAME: autoconfig - CHART_FOLDER: deploy/helm/autoconfig - REGISTRY_PROJECT: hardcore-tech - script: - - 'sed -i -E "s/^version:.*/version: ${CI_COMMIT_TAG}/; s/^appVersion:.*/appVersion: \"${CI_COMMIT_TAG}\"/" "${CHART_FOLDER}/Chart.yaml"' - - push-chart - - 'echo "chart pushed: autoconfig ${CI_COMMIT_TAG} → ChartMuseum(harbor-chart-repo)"' diff --git a/CHANGELOG.md b/CHANGELOG.md new file mode 100644 index 0000000..cc15eea --- /dev/null +++ b/CHANGELOG.md @@ -0,0 +1,90 @@ +# Changelog + +All notable changes to this project are documented here. The format follows +[Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and versions follow +[Semantic Versioning](https://semver.org/spec/v2.0.0.html). One version tag +releases the controller image and both sidecar images under the same number. + +## [Unreleased] + +### Added +- CI on every pull request: `gofmt`, `go vet`, `make test` with a check that + generated files are committed, `helm lint` of the chart, a docker build, and + a gate for the license text and committed credentials. +- Dependabot for Go modules and the GitHub Actions, which are pinned to commit + SHAs. +- `NOTICE`. +- English `README.md` and `DEPLOY.md`; the Chinese originals are kept as + `README.zh-CN.md` and `DEPLOY.zh-CN.md`. + +### Changed +- The Makefile, the kustomize manager config and the Helm chart default to + the public images (`4pdosc/autoconfig*`); `make docker-build` passes no + build-arg overrides by default. +- `DEPLOY.md` installs the charts from the public Helm repository + (`https://modelsphere.github.io/helm-charts`). +- The e2e scripts and samples use public images. `MON_IMG` has no public + default and must be set. + +### Removed +- The internal GitLab pipeline (`.gitlab-ci.yml`). + +### Fixed +- The arm64 images contained an amd64 binary: the Dockerfiles fixed + `GOARCH=amd64`. They now build for the target platform. + +## [0.4.0] - 2026-09-25 + +### Changed +- **Breaking:** `ModelRoute` moves from `routing.gpucluster.io` to + `routing.modelsphere.dev`, and the controller reads `LLMSLORequirement` from + `inference.modelsphere.dev` instead of `inference.x-k8s.io`. The finalizer + and the leader-election ID follow the routing group. Objects created under + the old groups are not picked up. + +## [0.3.47] - 2026-09-22 + +### Added +- Release workflow: a version tag builds `autoconfig`, `autoconfig-reload` + and `autoconfig-hagate` for linux/amd64 and linux/arm64 and pushes them to + Docker Hub (`4pdosc/`). + +## [0.3.46] - 2026-09-18 + +### Added +- The reload sidecar accepts `--watch` more than once, so changes in several + mounted volumes (for example a route ConfigMap and a key Secret) reload the + same process. One path behaves as before. +- Apache-2.0 `LICENSE`. + +### Changed +- Base images and the Go module proxy are Dockerfile build args that default + to public sources. +- README rewritten for readers outside the original team. +- The e2e scripts take the openresty auth key from `AUTH_KEY`, which must be + set. + +### Fixed +- The chart pinned `image.tag` to 0.3.45, so upgrading the chart kept the old + controller image. It is empty again and falls back to `appVersion`. + +## [0.3.45] - 2026-09-16 + +First tagged release. Earlier versions were built from this history without +tags. + +### Added +- `spec.cart.outputKey`: the ConfigMap key autoconfig writes CART workers to + (default `config.yaml`). Pointing it at a workers-only key leaves the base + config key to the chart. +- Already in place at this release: the `ModelRoute` CRD with CEL validation; + backend discovery from EndpointSlices or pod labels; rendering of openresty + routes (`llm` and `video` model types), CART workers and monitor targets into + existing ConfigMaps; the `autoconfig-reload` and `autoconfig-hagate` + sidecars; a Helm chart and kustomize manifests. + +[Unreleased]: https://github.com/modelsphere/autoconfig/compare/0.4.0...HEAD +[0.4.0]: https://github.com/modelsphere/autoconfig/compare/0.3.47...0.4.0 +[0.3.47]: https://github.com/modelsphere/autoconfig/compare/0.3.46...0.3.47 +[0.3.46]: https://github.com/modelsphere/autoconfig/compare/0.3.45...0.3.46 +[0.3.45]: https://github.com/modelsphere/autoconfig/releases/tag/0.3.45 diff --git a/DEPLOY.md b/DEPLOY.md index decde66..28d4953 100644 --- a/DEPLOY.md +++ b/DEPLOY.md @@ -1,280 +1,337 @@ -# k8s 部署手册(autoconfig 路由栈 + 测试模型 + ModelRoute) +# Kubernetes deployment guide (autoconfig routing stack + test model + ModelRoute) -在一套干净的 k8s 集群上,把整套「ModelRoute 驱动的 LLM 路由栈」部署起来,并跑通一个测试模型(以 **qwen** 为例;opt 等其它模型同样方式接入)。 -所有组件镜像 + helm chart 都由各仓 CI 打 git tag 后产出到 harbor / ChartMuseum,本文档只做 `helm install` + `kubectl apply`。 +English | [简体中文](DEPLOY.zh-CN.md) -## 0. 架构与依赖顺序 +This guide deploys the whole "ModelRoute-driven LLM routing stack" on a clean Kubernetes cluster and brings up +one test model end to end (**qwen** as the example; opt and other models are added the same way). +Component images are published to Docker Hub (`4pdosc/*`) when each repository is tagged, and the Helm charts +are published from [modelsphere/helm-charts](https://github.com/modelsphere/helm-charts); this guide only runs +`helm install` and `kubectl apply`. + +## 0. Architecture and dependency order ``` - ┌── autoconfig(operator)──┐ watch ModelRoute + EndpointSlice - ModelRoute(CR) ─▶│ 渲染 3 个 ConfigMap: │ → openresty-conf / cart--config / monitor-conf + ┌── autoconfig (operator) ──┐ watches ModelRoute + EndpointSlice + ModelRoute (CR) ▶│ renders 3 ConfigMaps: │ → openresty-conf / cart--config / monitor-conf └──────────┬────────────────┘ ┌──────────────┬───────┴───────┬──────────────┐ - openresty cart- monitor (各自挂对应 ConfigMap,reload sidecar 热更) - (对外入口 8080) (cache 亲和路由) (dashboard) + openresty cart- monitor (each mounts its ConfigMap; the reload sidecar hot-reloads) + (entry, 8080) (cache-affinity (dashboard) + routing) │ │ - └── 3 层 peer:cart(优先)→ backend pod-IP → backend-svc VIP 兜底 ──▶ 模型后端(vllm/sglang) + └── 3 peer tiers: cart (preferred) → backend pod IPs → backend-svc VIP fallback ──▶ model backend (vllm/sglang) ``` -**部署顺序(有依赖,别颠倒)**: -1. 前置:helm repo(ChartMuseum)+ namespace -2. **autoconfig**(先装 —— 它带 ModelRoute CRD + controller;后面组件都消费它产出的 ConfigMap) -3. 测试模型后端(qwen) -4. **cart**(每模型一个;`waitForWorkers` 会 Init 等 autoconfig 写入 workers) -5. **openresty**(对外入口) -6. **monitor**(dashboard + MySQL) -7. **ModelRoute CR**(qwen)→ autoconfig 据此填三个 ConfigMap → cart 就绪、openresty 出路由、monitor 出监控 -8. 验证 +**Deployment order (there are dependencies; do not reorder)**: +1. Prerequisites: Helm repository + namespaces +2. **autoconfig** (first — it brings the ModelRoute CRD and the controller; everything after consumes the + ConfigMaps it produces) +3. Test model backend (qwen) +4. **cart** (one per model; with `waitForWorkers` it waits in Init until autoconfig writes the workers) +5. **openresty** (the entry point) +6. **monitor** (dashboard + MySQL) +7. **ModelRoute CR** (qwen) → autoconfig fills the three ConfigMaps → cart becomes ready, openresty gets the + route, monitor starts monitoring it +8. Verify + +Images come from Docker Hub (`4pdosc/`) by default; charts come from the Helm repository +`https://modelsphere.github.io/helm-charts`. If the cluster has no internet access, mirror the images into your +own registry and override each chart's `image` values. -镜像统一在 `harbor.4pd.io/hardcore-tech/`;chart 统一在 ChartMuseum `https://harbor.4pd.io/chartrepo/hardcore-tech`。 +monitor has no public chart or image yet: in step 6, substitute your own monitor chart, or skip it (see step 6). --- -## 1. 前置 +## 1. Prerequisites ```bash -# 1.1 helm 加 ChartMuseum 仓(harbor 的 hardcore-tech project 允许匿名 pull → 只读无需凭证) -helm repo add harbor-chart-repo https://harbor.4pd.io/chartrepo/hardcore-tech -helm repo update harbor-chart-repo +# 1.1 Add the Helm repository (public, read-only, no credentials) +helm repo add modelsphere https://modelsphere.github.io/helm-charts +helm repo update modelsphere -# 1.2 确认能看到各 chart(注:helm search 对本 ChartMuseum 偶发空,用 helm show chart 确认;不加 --version 默认取最新) -helm show chart harbor-chart-repo/autoconfig | grep -E '^name|^version' -helm show chart harbor-chart-repo/openresty | grep -E '^name|^version' -helm show chart harbor-chart-repo/cache_aware_router | grep -E '^name|^version' -helm show chart harbor-chart-repo/monitor | grep -E '^name|^version' +# 1.2 Check that the charts are visible (without --version, the latest is used) +helm search repo modelsphere/autoconfig +helm search repo modelsphere/openresty +helm search repo modelsphere/cart -# 1.3 namespace(qwen ns 由后端 sample 自带 Namespace,无需先建) +# 1.3 Namespaces (the qwen namespace comes with the backend sample; no need to create it) kubectl create ns llm-route 2>/dev/null || true kubectl create ns monitoring 2>/dev/null || true ``` -> **本文档所有 `helm install`/`upgrade` 都不带 `--version`** —— helm 默认拉 ChartMuseum 里的最新版本,新 tag 一发布下次执行就自动用上,无需改本文档维护版本号。若需要锁定/回滚到某个历史版本,再显式加 `--version `(可用版本:`curl -s https://harbor.4pd.io/api/chartrepo/hardcore-tech/charts/`)。 +> **No `helm install`/`upgrade` in this guide passes `--version`** — Helm takes the latest version in the +> repository, so a new release is picked up the next time you run the command without editing this guide. To +> pin or roll back to an earlier version, add `--version ` (available versions: +> `helm search repo modelsphere/ --versions`). --- -## 2. autoconfig(operator + CRD) +## 2. autoconfig (operator + CRD) -chart 自带 `crds/`(ModelRoute CRD)+ controller Deployment(2 副本 leader 选举)+ RBAC。 +The chart ships `crds/` (the ModelRoute CRD), the controller Deployment (2 replicas with leader election) and +RBAC. ```bash -helm -n llm-route install autoconfig harbor-chart-repo/autoconfig \ +helm -n llm-route install autoconfig modelsphere/autoconfig \ --set fullnameOverride=autoconfig-controller -# 校验:CRD 装上 + controller Running +# Check: CRD installed + controller running kubectl get crd modelroutes.routing.modelsphere.dev kubectl -n llm-route rollout status deploy/autoconfig-controller ``` -### 2.1 健康探针与指标(0.3.32 起) +### 2.1 Health probes and metrics (since 0.3.32) -容器暴露两个端口,都**只给 k8s / Prometheus 用**,不承载业务: +The container exposes two ports, **for Kubernetes and Prometheus only**; they carry no application traffic: -| 端口 | 路径 | 用途 | +| Port | Path | Purpose | |---|---|---| -| 8081 | `/healthz` `/readyz` | liveness / readiness 探针 | -| 8080 | `/metrics` | controller-runtime 自带指标(Prometheus 抓) | +| 8081 | `/healthz` `/readyz` | liveness / readiness probes | +| 8080 | `/metrics` | controller-runtime's built-in metrics (scraped by Prometheus) | ```bash -# 校验探针 +# Check the probes kubectl -n llm-route get deploy autoconfig-controller \ -o jsonpath='{.spec.template.spec.containers[0].livenessProbe.httpGet}{"\n"}' -# 校验 ServiceMonitor 被 Prometheus 收编(注意 label release=kube-prometheus-stack) +# Check that Prometheus picks up the ServiceMonitor (note the label release=kube-prometheus-stack) kubectl -n llm-route get servicemonitor autoconfig-controller -o jsonpath='{.metadata.labels}{"\n"}' -# 抓到没:Prometheus 里应有 job=autoconfig-controller-metrics 的 target +# Is it scraped? Prometheus should have a target with job=autoconfig-controller-metrics ``` -**⚠️ 两个副本是主备,但 Service 里两个都在**。k8s Service 只按 label 选 pod,**不认 leader**; -autoconfig 也没有 hagate 侧车(它不接流量,不需要把 standby 摘出 endpoints)。 -所以 metrics Service 做成 **headless(`clusterIP: None`)**,让 Prometheus 按 pod 逐个抓、指标带 `pod` 标签分开。 -**直接 curl Service 会随机落到某个副本,可能是 standby,看到的队列恒空、reconcile 计数近 0,别误判成没干活。** +**⚠️ The two replicas are master-standby, but both are in the Service.** A Kubernetes Service selects pods by +label only and **knows nothing about the leader**; autoconfig has no hagate sidecar either (it takes no traffic, +so there is no need to take the standby out of the endpoints). The metrics Service is therefore **headless +(`clusterIP: None`)**, so Prometheus scrapes each pod and the metrics carry a `pod` label. +**A plain curl to the Service lands on a random replica, possibly the standby, whose queue is always empty and +whose reconcile count is near zero — do not mistake that for the controller doing nothing.** -在 Prometheus 里认 leader 用 `leader_election_master_status`(leader=1 / standby=0,controller-runtime 自带): +To identify the leader in Prometheus, use `leader_election_master_status` (leader=1 / standby=0, built into +controller-runtime): ```promql -# 只看 leader 的队列积压 +# The leader's queue backlog only workqueue_depth{job=~"autoconfig.*"} and on(pod) (leader_election_master_status == 1) ``` -**排查 reconcile 卡死(如 cart ConfigMap wedge)看这条**——健康探针发现不了,它只证明进程能应答 HTTP: +**To diagnose a stuck reconcile (such as a wedged cart ConfigMap), look at this one** — the health probes cannot +catch it; they only prove the process answers HTTP: ```promql -# 当前这次 reconcile 已经跑了多久;持续上涨且不归零 = 卡住了 +# How long the current reconcile has been running; rising without returning to zero = stuck workqueue_unfinished_work_seconds{job=~"autoconfig.*"} and on(pod) (leader_election_master_status == 1) ``` -其余常用:`workqueue_depth`(排队的 ModelRoute 数,线上 4 个对象 + 10s resync,稳态 0~1)、 -`workqueue_retries_total`(调谐失败重试,DiscoverError / 底稿读空会陡增)、 -`controller_runtime_reconcile_errors_total`、`rest_client_requests_total`(出现 429 = 被 apiserver 限流)、 -`go_goroutines`(泄漏)。 +Other useful ones: `workqueue_depth` (ModelRoutes waiting in the queue; with a handful of objects and the 10s +resync, 0–1 in steady state), `workqueue_retries_total` (reconcile retries; jumps on DiscoverError or an empty +base config), `controller_runtime_reconcile_errors_total`, `rest_client_requests_total` (429s = throttled by the +API server), `go_goroutines` (leaks). -leader 是 k8s Lease,查当前持有者: +The leader is a Kubernetes Lease; to see the current holder: ```bash kubectl -n llm-route get lease autoconfig-controller.routing.modelsphere.dev -o jsonpath='{.spec.holderIdentity}{"\n"}' ``` -关掉指标(如不想被抓):`--set metrics.enabled=false` 或 `--set metrics.serviceMonitor.enabled=false`。 +To turn metrics off (if you do not want them scraped): `--set metrics.enabled=false` or +`--set metrics.serviceMonitor.enabled=false`. --- -## 3. 测试模型后端(以 qwen 为例) +## 3. Test model backend (qwen as the example) -sample 在 `config/samples/`,自带 Namespace + Deployment + Service。 +The samples are in `config/samples/` and include the Namespace, Deployment and Service. -- `qwen-backend.yaml`:sglang `Qwen3.5-4B`(ns `qwen`,`qwen-svc` NodePort:30055;**`--enable-metrics`** 否则 monitor 采不到 KV/running/waiting;`terminationGracePeriodSeconds:3600` 排空长请求) +- `qwen-backend.yaml`: sglang `Qwen3.5-4B` (namespace `qwen`, `qwen-svc` NodePort 30055; **`--enable-metrics`**, + without which the monitor cannot collect KV/running/waiting; `terminationGracePeriodSeconds: 3600` to drain long + requests). The model is read from a `hostPath`; set `nodeName` (commented out in the sample) to the node that + holds it. ```bash kubectl apply -f config/samples/qwen-backend.yaml kubectl -n qwen rollout status deploy/qwen -# sglang /metrics 需 --enable-metrics 才 200(默认 404): +# sglang /metrics returns 200 only with --enable-metrics (404 by default): kubectl -n qwen exec deploy/qwen -- sh -c "curl -s -o /dev/null -w '%{http_code}\n' http://127.0.0.1:8000/metrics" ``` -> **opt 同理**:`config/samples/opt-backend.yaml`(vLLM `opt-125m`,ns `opt`,`opt-svc` ClusterIP:8000;`--shutdown-timeout=3540` 优雅停机)+ `modelroute-opt.yaml`,后续步骤把 `qwen` 换成 `opt` 即可。 -> 生产模型(如 kimi 用 LeaderWorkerSet TP8/PP2)部署方式不同(见 `scripts/k8s-llm/`),但接入路由的方式一样:建 Service + 写 ModelRoute。 +> **opt works the same way**: `config/samples/opt-backend.yaml` (vLLM `opt-125m`, namespace `opt`, `opt-svc` +> ClusterIP 8000; `--shutdown-timeout=3540` for a graceful shutdown) + `modelroute-opt.yaml`; in the following +> steps, replace `qwen` with `opt`. +> Production models (for example multi-node deployments with LeaderWorkerSet) are deployed differently (see the +> `sglang` / `vllm` charts in modelsphere/helm-charts), but they join the routing the same way: create a Service +> and write a ModelRoute. --- -## 4. cart(cache_aware_router,每模型一个) +## 4. cart (cache_aware_router, one per model) -**cart 是「一个模型一个」**(radix 前缀缓存只对单模型有效)。`fullnameOverride` 决定 Service 名 + config ConfigMap 名(`-config`),ModelRoute 的 `cart.service` / `cart.outputConfigMap` 要对上。 -`waitForWorkers=true`:cart pod 停在 Init 等 autoconfig 写 workers(第 7 步 apply ModelRoute 后就绪),不会 CrashLoop。 +**There is one cart per model** (the radix prefix cache only helps within one model). `fullnameOverride` decides +the Service name and the config ConfigMap name (`-config`); the ModelRoute's `cart.service` / +`cart.outputConfigMap` must match them. +With `waitForWorkers=true`, the cart pod waits in Init for autoconfig to write the workers (ready after the +ModelRoute is applied in step 7) instead of crash-looping. ```bash -# reload/hagate 侧车镜像版本已是 chart 的默认值(每次 autoconfig 发版会同步 bump 进各 chart), -# 不用显式 --set 覆盖 —— 显式写死版本号反而会在 chart 默认值升级后仍锁在旧版,忘了改就悄悄漂移。 -helm -n llm-route install cart-qwen harbor-chart-repo/cache_aware_router \ +# The reload/hagate sidecar images are already the chart's defaults (bumped in each chart when autoconfig +# releases), so do not override them with --set — a pinned version stays pinned after the chart default moves, +# and drifts silently if you forget to update it. +helm -n llm-route install cart-qwen modelsphere/cart \ --set fullnameOverride=cart-qwen -# 此时 cart pod 会停在 Init(等 workers),属正常;第 7 步后转 Running +# The cart pod now waits in Init (for workers); that is expected. It turns Running after step 7 kubectl -n llm-route get pods | grep cart- ``` -> opt 同理:`--set fullnameOverride=cart-opt`(要与 `modelroute-opt.yaml` 的 `cart.service`/`cart.outputConfigMap` 对上)。 +> opt works the same way: `--set fullnameOverride=cart-opt` (must match `cart.service`/`cart.outputConfigMap` +> in `modelroute-opt.yaml`). --- -## 5. openresty(对外入口) +## 5. openresty (the entry point) -chart 挂载 autoconfig 产出的 `openresty-conf` ConfigMap(各 `session_route_.conf`),reload sidecar 收 SIGHUP 热更路由。 -`image.tag` 用 chart 默认(= chart 的 `appVersion`,随最新 tag 走);`reload.image`/`ha.image` 同理用 chart 默认,不显式覆盖; -`bodylog.host` 指向 bodylog-listener(k8s 里需 FQDN 或可解析地址)。 +The chart mounts the `openresty-conf` ConfigMap autoconfig produces (one `session_route_.conf` per model), +and the reload sidecar hot-reloads the routes on SIGHUP. +`image.tag` uses the chart default (= the chart's `appVersion`, following the latest release); likewise leave +`reload.image`/`ha.image` at the chart defaults. `bodylog.host` points at the bodylog listener (in Kubernetes it +must be an FQDN or another resolvable address). ```bash -helm -n llm-route install openresty harbor-chart-repo/openresty \ +helm -n llm-route install openresty modelsphere/openresty \ --set fullnameOverride=openresty \ --set bodylog.host=192.0.2.31 kubectl -n llm-route rollout status deploy/openresty ``` -> openresty Service 是 **ClusterIP:8080**(集群内访问);对外测试用 `kubectl -n llm-route port-forward svc/openresty 18080:8080`。 +> The openresty Service is **ClusterIP 8080** (in-cluster access); to test from outside, use +> `kubectl -n llm-route port-forward svc/openresty 18080:8080`. --- -## 6. monitor(dashboard + MySQL) +## 6. monitor (dashboard + MySQL) + +The chart includes MySQL (persistent state + time series) and reads the `monitor-conf` autoconfig produces +(service/nginx/router lines, hot-reloaded every 60s). -chart 自带 MySQL(持久化 state + 时序);读 autoconfig 产出的 `monitor-conf`(service/nginx/router 行,60s 热加载)。 +> monitor has no public chart: replace `` below with your own chart source. Without a monitor, +> skip this step and remove the `spec.monitor` block from the ModelRoute in step 7 (it is optional). -**生产推荐:密钥走 `existingSecret`(带外建,不归 helm 管)** —— 这样 `helm upgrade` 无论带不带 `--set` 都碰不到密钥,避免「误用 `--set` 不带 `--reuse-values` → 密钥被刷成占位」的坑(app + mysql 两份密钥都能外置)。 +**Recommended for production: keep secrets in an `existingSecret` (created out of band, not managed by Helm)** — +then `helm upgrade` never touches the secrets, with or without `--set`, which avoids the trap of "`--set` without +`--reuse-values` → secrets reset to placeholders" (both the app and the MySQL secrets can be external). ```bash -# ① 带外建两份 secret(模板见 monitor 仓 k8s/secret.example.yaml,改成真值)——注意 mysql 密码两处要一致 +# ① Create the two secrets out of band (template: k8s/secret.example.yaml in the monitor repository; fill in real values) — the MySQL password must match in both kubectl -n monitoring apply -f secret.example.yaml # llm-monitor-secret + llm-monitor-mysql-secret -# ② install:关掉 chart 自建密钥,引用带外的 -# (nginxHost / bodylogSummaryURL 已是 chart 默认值——bodylog 默认指 ts34,换集群才 --set 覆盖) -helm -n monitoring install monitor harbor-chart-repo/monitor \ +# ② Install: disable the chart's own secret and reference the external ones +# (override nginxHost / bodylogSummaryURL with --set for your cluster) +helm -n monitoring install monitor \ --set secret.create=false \ --set secret.existingSecret=llm-monitor-secret \ --set mysql.auth.existingSecret=llm-monitor-mysql-secret kubectl -n monitoring rollout status deploy/monitor ``` -> **快速起(测试用,chart 自建密钥)**:不想带外建 secret 时,可用 values 文件把密钥/密码写进去(`secret.data.*` + `mysql.auth.*`,占位改真值,**别提交 repo**)、`create:true` 装。但这种模式下升级务必 `--reuse-values`,且带 `--set` 时尤其小心(见 §9)。 +> **Quick start (for testing, chart-managed secrets)**: if you do not want external secrets, put the keys and +> passwords in a values file (`secret.data.*` + `mysql.auth.*`, replacing the placeholders; **never commit it**) +> and install with `create: true`. In this mode always upgrade with `--reuse-values`, and take particular care when +> passing `--set` (see §9). -> dashboard 是 **NodePort:30080** → `http://<任一 node IP>:30080`(如 `http://192.0.2.20:30080`),Basic Auth `admin/`;`/tpm` 子页独立 Auth `tpm/`。 +> The dashboard is **NodePort 30080** → `http://:30080` (for example `http://192.0.2.20:30080`), Basic +> Auth `admin/`; the `/tpm` page has its own auth `tpm/`. --- -## 7. 安装 ModelRoute(触发全栈自动配置) +## 7. Install the ModelRoute (triggers configuration of the whole stack) -ModelRoute 是**中心配置**:autoconfig 据此同时写 openresty-conf / cart--config / monitor-conf。 -sample 在 `config/samples/`,`modelroute-qwen.yaml`:qwen 三层路由(cart-qwen → backend pod-IP → backend-svc VIP 兜底);discovery `qwen/qwen-svc`;monitor model `qwen`。 +The ModelRoute is the **central configuration**: from it, autoconfig writes openresty-conf / +cart--config / monitor-conf at the same time. +The sample is `config/samples/modelroute-qwen.yaml`: three-tier routing for qwen (cart-qwen → backend pod IPs → +backend-svc VIP fallback); discovery `qwen/qwen-svc`; monitor model `qwen`. ```bash kubectl apply -f config/samples/modelroute-qwen.yaml -# autoconfig 会在 ~秒级 reconcile;ConfigMap→pod 挂载传播有 ~1min kubelet 同步 lag -kubectl -n llm-route get modelroute qwen # READY 应为 true,BACKENDS≥1,CART=1 +# autoconfig reconciles within seconds; ConfigMap → pod mount propagation lags by ~1 min (kubelet sync) +kubectl -n llm-route get modelroute qwen # READY should be true, BACKENDS ≥ 1, CART = 1 ``` -apply 后应观察到:cart-qwen pod 从 Init 转 **Running**(autoconfig 写入 workers);openresty 出现 `session_route_qwen.conf`;monitor dashboard 出现 qwen 的 service 行。 +After applying, you should see: the cart-qwen pod go from Init to **Running** (autoconfig wrote the workers); +`session_route_qwen.conf` appear in openresty; and qwen's service line appear on the monitor dashboard. -> **ModelRoute 放哪个 ns?** 本例放 `llm-route`(与 cart/openresty 同 ns,sample 里 `cart.service: cart-qwen`、`nginx.service: openresty` 用裸名即可)。controller 是全集群 watch(ClusterRole),ModelRoute 放 model ns(如 `qwen`)也行,但那些**裸引用会默认解析到 ModelRoute 自己的 ns** → 需显式加前缀 `llm-route/cart-qwen`、`llm-route/openresty`(各 `outputConfigMap` 本就带 ns,不用改)。 -> **opt 同理**:`kubectl apply -f config/samples/modelroute-opt.yaml`(discovery `opt/opt-svc`、cart-opt、monitor model `opt-125m`)。 +> **Which namespace for the ModelRoute?** This example uses `llm-route` (the same namespace as cart/openresty, so +> the sample's `cart.service: cart-qwen` and `nginx.service: openresty` can be bare names). The controller watches +> the whole cluster (ClusterRole), so the ModelRoute can also live in the model's namespace (such as `qwen`), but +> then **bare references resolve to the ModelRoute's own namespace** → prefix them explicitly: +> `llm-route/cart-qwen`, `llm-route/openresty` (each `outputConfigMap` already includes its namespace). +> **opt works the same way**: `kubectl apply -f config/samples/modelroute-opt.yaml` (discovery `opt/opt-svc`, +> cart-opt, monitor model `opt-125m`). --- -## 8. 验证(端到端) +## 8. Verify (end to end) ```bash -# 8.1 ModelRoute 就绪 +# 8.1 ModelRoute ready kubectl -n llm-route get modelroute -# 8.2 cart 就绪(3/3,含 cart + reload + hagate 侧车) +# 8.2 cart ready (3/3: cart + reload + hagate sidecars) kubectl -n llm-route get pods | grep -E 'cart-|openresty' -# 8.3 端到端经 openresty → cart → 后端(在 openresty pod 内打本地 8080) -AUTH_KEY='' # 真实 key 不入库 +# 8.3 End to end through openresty → cart → backend (call local port 8080 inside the openresty pod) +AUTH_KEY='' # never commit the real key ORP=$(kubectl -n llm-route get pod -l app.kubernetes.io/name=openresty -o jsonpath='{.items[0].metadata.name}' 2>/dev/null) [ -z "$ORP" ] && ORP=$(kubectl -n llm-route get pods -o name | grep openresty | head -1 | cut -d/ -f2) kubectl -n llm-route exec $ORP -c openresty -- sh -c \ "curl -s -o /dev/null -w 'qwen /v1/models=%{http_code}\n' http://127.0.0.1:8080/qwen/v1/models -H 'Authorization: Bearer $AUTH_KEY'" -# chat 流(应 200,并回 X-Routed-Peer 头): +# A chat stream (should be 200, with an X-Routed-Peer header): kubectl -n llm-route exec $ORP -c openresty -- sh -c \ "curl -s -D - -o /dev/null http://127.0.0.1:8080/qwen/v1/chat/completions -H 'Content-Type: application/json' \ -H 'Authorization: Bearer $AUTH_KEY' \ -d '{\"model\":\"qwen\",\"messages\":[{\"role\":\"user\",\"content\":\"hi\"}],\"max_tokens\":8}' | grep -iE 'HTTP/|x-routed-peer'" -# 8.4 monitor 采到该模型(dashboard 或 /api/status) +# 8.4 The monitor collects the model (dashboard or /api/status) kubectl -n monitoring exec deploy/monitor -- python3 -c \ 'import urllib.request,base64,json;r=urllib.request.Request("http://127.0.0.1:8080/api/status");r.add_header("Authorization","Basic "+base64.b64encode(b"admin:").decode());print("api/status",json.load(urllib.request.urlopen(r)) and "OK")' ``` -`X-Routed-Peer` 头出现 = 请求确实经 cart 路由到了真实后端(cart 的 `proxy.add_routed_peer_header=true`)。 +An `X-Routed-Peer` header means the request really was routed by cart to a real backend (cart's +`proxy.add_routed_peer_header=true`). --- -## 9. 升级 / 回滚(chart 已在 ChartMuseum) +## 9. Upgrade / roll back -发新版流程:改代码 → 各仓打 git tag(CI 自动出镜像 + push chart 到 ChartMuseum)→ 集群 `helm upgrade`。 +Releasing a new version: change the code → tag each repository (CI publishes the images to Docker Hub) → the +charts are released in modelsphere/helm-charts → `helm upgrade` in the cluster. ```bash -helm repo update harbor-chart-repo -# 不带 --version = 拉最新 tag;--reuse-values 保留安装时的 fullnameOverride / 密钥 / bodylog.host 等 -helm -n llm-route upgrade autoconfig harbor-chart-repo/autoconfig --reuse-values -helm -n llm-route upgrade openresty harbor-chart-repo/openresty --reuse-values -helm -n llm-route upgrade cart-qwen harbor-chart-repo/cache_aware_router --reuse-values # 每个 cart- 各升一次 -helm -n monitoring upgrade monitor harbor-chart-repo/monitor --reuse-values - -helm -n history # 看修订 -helm -n rollback # 回滚到某修订 -helm -n upgrade --version --reuse-values # 需要锁定/回退到某个历史版本才加 --version +helm repo update modelsphere +# Without --version = the latest release; --reuse-values keeps fullnameOverride / secrets / bodylog.host etc. from install time +helm -n llm-route upgrade autoconfig modelsphere/autoconfig --reuse-values +helm -n llm-route upgrade openresty modelsphere/openresty --reuse-values +helm -n llm-route upgrade cart-qwen modelsphere/cart --reuse-values # once for each cart- +helm -n monitoring upgrade monitor --reuse-values + +helm -n history # list revisions +helm -n rollback # roll back to a revision +helm -n upgrade --version --reuse-values # add --version only to pin or go back to an earlier version ``` -> chart 内容不变时,`helm upgrade` 只更新 release 元数据、**不重启 pod**(渲染出的 spec 一致),零中断。 -> chart 版本命名:多数仓 = git tag(0.1.x / 0.3.x);**cache_aware_router 例外** —— git tag 带前导 `v`(如 `v0.6.2-k8s`),chart version 去掉 `v`(`0.6.2-k8s`,SemVer2 不许带 v),`--version` 用去 v 的。 +> When the chart content is unchanged, `helm upgrade` only updates the release metadata and **does not restart +> pods** (the rendered spec is identical): no interruption. +> A chart version is not necessarily the component version (for example the `cart` chart 0.2.x deploys CART +> 0.6.x); to pin a version, look up the chart versions with `helm search repo modelsphere/ --versions`. --- -## 10. 卸载 / 清理 +## 10. Uninstall / clean up ```bash kubectl delete -f config/samples/modelroute-qwen.yaml -helm -n llm-route uninstall openresty cart-qwen autoconfig # opt 同理:再 uninstall cart-opt +helm -n llm-route uninstall openresty cart-qwen autoconfig # opt: also uninstall cart-opt helm -n monitoring uninstall monitor -kubectl delete -f config/samples/qwen-backend.yaml # 连带删 qwen ns(opt 同理删 opt-backend.yaml) -kubectl delete crd modelroutes.routing.modelsphere.dev # 如需彻底移除 CRD +kubectl delete -f config/samples/qwen-backend.yaml # also deletes the qwen namespace (opt: opt-backend.yaml) +kubectl delete crd modelroutes.routing.modelsphere.dev # to remove the CRD completely ``` diff --git a/DEPLOY.zh-CN.md b/DEPLOY.zh-CN.md new file mode 100644 index 0000000..7f99c6f --- /dev/null +++ b/DEPLOY.zh-CN.md @@ -0,0 +1,286 @@ +# k8s 部署手册(autoconfig 路由栈 + 测试模型 + ModelRoute) + +[English](DEPLOY.md) | 简体中文 + +在一套干净的 k8s 集群上,把整套「ModelRoute 驱动的 LLM 路由栈」部署起来,并跑通一个测试模型(以 **qwen** 为例;opt 等其它模型同样方式接入)。 +组件镜像由各仓打 git tag 后发布到 Docker Hub(`4pdosc/*`),helm chart 发布在 [modelsphere/helm-charts](https://github.com/modelsphere/helm-charts);本文档只做 `helm install` + `kubectl apply`。 + +## 0. 架构与依赖顺序 + +``` + ┌── autoconfig(operator)──┐ watch ModelRoute + EndpointSlice + ModelRoute(CR) ─▶│ 渲染 3 个 ConfigMap: │ → openresty-conf / cart--config / monitor-conf + └──────────┬────────────────┘ + ┌──────────────┬───────┴───────┬──────────────┐ + openresty cart- monitor (各自挂对应 ConfigMap,reload sidecar 热更) + (对外入口 8080) (cache 亲和路由) (dashboard) + │ │ + └── 3 层 peer:cart(优先)→ backend pod-IP → backend-svc VIP 兜底 ──▶ 模型后端(vllm/sglang) +``` + +**部署顺序(有依赖,别颠倒)**: +1. 前置:helm repo + namespace +2. **autoconfig**(先装 —— 它带 ModelRoute CRD + controller;后面组件都消费它产出的 ConfigMap) +3. 测试模型后端(qwen) +4. **cart**(每模型一个;`waitForWorkers` 会 Init 等 autoconfig 写入 workers) +5. **openresty**(对外入口) +6. **monitor**(dashboard + MySQL) +7. **ModelRoute CR**(qwen)→ autoconfig 据此填三个 ConfigMap → cart 就绪、openresty 出路由、monitor 出监控 +8. 验证 + +镜像默认来自 Docker Hub(`4pdosc/`);chart 来自 helm 仓 `https://modelsphere.github.io/helm-charts`。 +集群不能访问外网时,把镜像同步到自己的镜像仓库,再用各 chart 的 `image` 相关 values 覆盖。 + +monitor 目前没有公开的 chart 和镜像:第 6 步换成你自己的 monitor chart 来源,或者跳过它(见第 6 步)。 + +--- + +## 1. 前置 + +```bash +# 1.1 加 helm 仓(公开,只读无需凭证) +helm repo add modelsphere https://modelsphere.github.io/helm-charts +helm repo update modelsphere + +# 1.2 确认能看到各 chart(不加 --version 默认取最新) +helm search repo modelsphere/autoconfig +helm search repo modelsphere/openresty +helm search repo modelsphere/cart + +# 1.3 namespace(qwen ns 由后端 sample 自带 Namespace,无需先建) +kubectl create ns llm-route 2>/dev/null || true +kubectl create ns monitoring 2>/dev/null || true +``` + +> **本文档所有 `helm install`/`upgrade` 都不带 `--version`** —— helm 默认拉仓里的最新版本,新版本一发布下次执行就自动用上,无需改本文档维护版本号。若需要锁定/回滚到某个历史版本,再显式加 `--version `(可用版本:`helm search repo modelsphere/ --versions`)。 + +--- + +## 2. autoconfig(operator + CRD) + +chart 自带 `crds/`(ModelRoute CRD)+ controller Deployment(2 副本 leader 选举)+ RBAC。 + +```bash +helm -n llm-route install autoconfig modelsphere/autoconfig \ + --set fullnameOverride=autoconfig-controller + +# 校验:CRD 装上 + controller Running +kubectl get crd modelroutes.routing.modelsphere.dev +kubectl -n llm-route rollout status deploy/autoconfig-controller +``` + +### 2.1 健康探针与指标(0.3.32 起) + +容器暴露两个端口,都**只给 k8s / Prometheus 用**,不承载业务: + +| 端口 | 路径 | 用途 | +|---|---|---| +| 8081 | `/healthz` `/readyz` | liveness / readiness 探针 | +| 8080 | `/metrics` | controller-runtime 自带指标(Prometheus 抓) | + +```bash +# 校验探针 +kubectl -n llm-route get deploy autoconfig-controller \ + -o jsonpath='{.spec.template.spec.containers[0].livenessProbe.httpGet}{"\n"}' + +# 校验 ServiceMonitor 被 Prometheus 收编(注意 label release=kube-prometheus-stack) +kubectl -n llm-route get servicemonitor autoconfig-controller -o jsonpath='{.metadata.labels}{"\n"}' +# 抓到没:Prometheus 里应有 job=autoconfig-controller-metrics 的 target +``` + +**⚠️ 两个副本是主备,但 Service 里两个都在**。k8s Service 只按 label 选 pod,**不认 leader**; +autoconfig 也没有 hagate 侧车(它不接流量,不需要把 standby 摘出 endpoints)。 +所以 metrics Service 做成 **headless(`clusterIP: None`)**,让 Prometheus 按 pod 逐个抓、指标带 `pod` 标签分开。 +**直接 curl Service 会随机落到某个副本,可能是 standby,看到的队列恒空、reconcile 计数近 0,别误判成没干活。** + +在 Prometheus 里认 leader 用 `leader_election_master_status`(leader=1 / standby=0,controller-runtime 自带): + +```promql +# 只看 leader 的队列积压 +workqueue_depth{job=~"autoconfig.*"} and on(pod) (leader_election_master_status == 1) +``` + +**排查 reconcile 卡死(如 cart ConfigMap wedge)看这条**——健康探针发现不了,它只证明进程能应答 HTTP: + +```promql +# 当前这次 reconcile 已经跑了多久;持续上涨且不归零 = 卡住了 +workqueue_unfinished_work_seconds{job=~"autoconfig.*"} and on(pod) (leader_election_master_status == 1) +``` + +其余常用:`workqueue_depth`(排队的 ModelRoute 数,线上 4 个对象 + 10s resync,稳态 0~1)、 +`workqueue_retries_total`(调谐失败重试,DiscoverError / 底稿读空会陡增)、 +`controller_runtime_reconcile_errors_total`、`rest_client_requests_total`(出现 429 = 被 apiserver 限流)、 +`go_goroutines`(泄漏)。 + +leader 是 k8s Lease,查当前持有者: + +```bash +kubectl -n llm-route get lease autoconfig-controller.routing.modelsphere.dev -o jsonpath='{.spec.holderIdentity}{"\n"}' +``` + +关掉指标(如不想被抓):`--set metrics.enabled=false` 或 `--set metrics.serviceMonitor.enabled=false`。 + +--- + +## 3. 测试模型后端(以 qwen 为例) + +sample 在 `config/samples/`,自带 Namespace + Deployment + Service。 + +- `qwen-backend.yaml`:sglang `Qwen3.5-4B`(ns `qwen`,`qwen-svc` NodePort:30055;**`--enable-metrics`** 否则 monitor 采不到 KV/running/waiting;`terminationGracePeriodSeconds:3600` 排空长请求)。模型从 `hostPath` 读取,把 sample 里注释掉的 `nodeName` 设成存放模型的节点 + +```bash +kubectl apply -f config/samples/qwen-backend.yaml +kubectl -n qwen rollout status deploy/qwen +# sglang /metrics 需 --enable-metrics 才 200(默认 404): +kubectl -n qwen exec deploy/qwen -- sh -c "curl -s -o /dev/null -w '%{http_code}\n' http://127.0.0.1:8000/metrics" +``` + +> **opt 同理**:`config/samples/opt-backend.yaml`(vLLM `opt-125m`,ns `opt`,`opt-svc` ClusterIP:8000;`--shutdown-timeout=3540` 优雅停机)+ `modelroute-opt.yaml`,后续步骤把 `qwen` 换成 `opt` 即可。 +> 生产模型(如 kimi 用 LeaderWorkerSet TP8/PP2)部署方式不同(见 modelsphere/helm-charts 的 `sglang` / `vllm` chart),但接入路由的方式一样:建 Service + 写 ModelRoute。 + +--- + +## 4. cart(cache_aware_router,每模型一个) + +**cart 是「一个模型一个」**(radix 前缀缓存只对单模型有效)。`fullnameOverride` 决定 Service 名 + config ConfigMap 名(`-config`),ModelRoute 的 `cart.service` / `cart.outputConfigMap` 要对上。 +`waitForWorkers=true`:cart pod 停在 Init 等 autoconfig 写 workers(第 7 步 apply ModelRoute 后就绪),不会 CrashLoop。 + +```bash +# reload/hagate 侧车镜像版本已是 chart 的默认值(每次 autoconfig 发版会同步 bump 进各 chart), +# 不用显式 --set 覆盖 —— 显式写死版本号反而会在 chart 默认值升级后仍锁在旧版,忘了改就悄悄漂移。 +helm -n llm-route install cart-qwen modelsphere/cart \ + --set fullnameOverride=cart-qwen + +# 此时 cart pod 会停在 Init(等 workers),属正常;第 7 步后转 Running +kubectl -n llm-route get pods | grep cart- +``` + +> opt 同理:`--set fullnameOverride=cart-opt`(要与 `modelroute-opt.yaml` 的 `cart.service`/`cart.outputConfigMap` 对上)。 + +--- + +## 5. openresty(对外入口) + +chart 挂载 autoconfig 产出的 `openresty-conf` ConfigMap(各 `session_route_.conf`),reload sidecar 收 SIGHUP 热更路由。 +`image.tag` 用 chart 默认(= chart 的 `appVersion`,随最新 tag 走);`reload.image`/`ha.image` 同理用 chart 默认,不显式覆盖; +`bodylog.host` 指向 bodylog-listener(k8s 里需 FQDN 或可解析地址)。 + +```bash +helm -n llm-route install openresty modelsphere/openresty \ + --set fullnameOverride=openresty \ + --set bodylog.host=192.0.2.31 + +kubectl -n llm-route rollout status deploy/openresty +``` + +> openresty Service 是 **ClusterIP:8080**(集群内访问);对外测试用 `kubectl -n llm-route port-forward svc/openresty 18080:8080`。 + +--- + +## 6. monitor(dashboard + MySQL) + +chart 自带 MySQL(持久化 state + 时序);读 autoconfig 产出的 `monitor-conf`(service/nginx/router 行,60s 热加载)。 + +> monitor 没有公开的 chart:下面的 `` 换成你自己的 chart 来源。不装 monitor 时跳过本步,并删掉第 7 步 ModelRoute 里的 `spec.monitor` 段(它是可选的)。 + +**生产推荐:密钥走 `existingSecret`(带外建,不归 helm 管)** —— 这样 `helm upgrade` 无论带不带 `--set` 都碰不到密钥,避免「误用 `--set` 不带 `--reuse-values` → 密钥被刷成占位」的坑(app + mysql 两份密钥都能外置)。 + +```bash +# ① 带外建两份 secret(模板见 monitor 仓 k8s/secret.example.yaml,改成真值)——注意 mysql 密码两处要一致 +kubectl -n monitoring apply -f secret.example.yaml # llm-monitor-secret + llm-monitor-mysql-secret + +# ② install:关掉 chart 自建密钥,引用带外的 +# (nginxHost / bodylogSummaryURL 按你的集群用 --set 覆盖) +helm -n monitoring install monitor \ + --set secret.create=false \ + --set secret.existingSecret=llm-monitor-secret \ + --set mysql.auth.existingSecret=llm-monitor-mysql-secret +kubectl -n monitoring rollout status deploy/monitor +``` + +> **快速起(测试用,chart 自建密钥)**:不想带外建 secret 时,可用 values 文件把密钥/密码写进去(`secret.data.*` + `mysql.auth.*`,占位改真值,**别提交 repo**)、`create:true` 装。但这种模式下升级务必 `--reuse-values`,且带 `--set` 时尤其小心(见 §9)。 + +> dashboard 是 **NodePort:30080** → `http://<任一 node IP>:30080`(如 `http://192.0.2.20:30080`),Basic Auth `admin/`;`/tpm` 子页独立 Auth `tpm/`。 + +--- + +## 7. 安装 ModelRoute(触发全栈自动配置) + +ModelRoute 是**中心配置**:autoconfig 据此同时写 openresty-conf / cart--config / monitor-conf。 +sample 在 `config/samples/`,`modelroute-qwen.yaml`:qwen 三层路由(cart-qwen → backend pod-IP → backend-svc VIP 兜底);discovery `qwen/qwen-svc`;monitor model `qwen`。 + +```bash +kubectl apply -f config/samples/modelroute-qwen.yaml + +# autoconfig 会在 ~秒级 reconcile;ConfigMap→pod 挂载传播有 ~1min kubelet 同步 lag +kubectl -n llm-route get modelroute qwen # READY 应为 true,BACKENDS≥1,CART=1 +``` + +apply 后应观察到:cart-qwen pod 从 Init 转 **Running**(autoconfig 写入 workers);openresty 出现 `session_route_qwen.conf`;monitor dashboard 出现 qwen 的 service 行。 + +> **ModelRoute 放哪个 ns?** 本例放 `llm-route`(与 cart/openresty 同 ns,sample 里 `cart.service: cart-qwen`、`nginx.service: openresty` 用裸名即可)。controller 是全集群 watch(ClusterRole),ModelRoute 放 model ns(如 `qwen`)也行,但那些**裸引用会默认解析到 ModelRoute 自己的 ns** → 需显式加前缀 `llm-route/cart-qwen`、`llm-route/openresty`(各 `outputConfigMap` 本就带 ns,不用改)。 +> **opt 同理**:`kubectl apply -f config/samples/modelroute-opt.yaml`(discovery `opt/opt-svc`、cart-opt、monitor model `opt-125m`)。 + +--- + +## 8. 验证(端到端) + +```bash +# 8.1 ModelRoute 就绪 +kubectl -n llm-route get modelroute + +# 8.2 cart 就绪(3/3,含 cart + reload + hagate 侧车) +kubectl -n llm-route get pods | grep -E 'cart-|openresty' + +# 8.3 端到端经 openresty → cart → 后端(在 openresty pod 内打本地 8080) +AUTH_KEY='' # 真实 key 不入库 +ORP=$(kubectl -n llm-route get pod -l app.kubernetes.io/name=openresty -o jsonpath='{.items[0].metadata.name}' 2>/dev/null) +[ -z "$ORP" ] && ORP=$(kubectl -n llm-route get pods -o name | grep openresty | head -1 | cut -d/ -f2) +kubectl -n llm-route exec $ORP -c openresty -- sh -c \ + "curl -s -o /dev/null -w 'qwen /v1/models=%{http_code}\n' http://127.0.0.1:8080/qwen/v1/models -H 'Authorization: Bearer $AUTH_KEY'" +# chat 流(应 200,并回 X-Routed-Peer 头): +kubectl -n llm-route exec $ORP -c openresty -- sh -c \ + "curl -s -D - -o /dev/null http://127.0.0.1:8080/qwen/v1/chat/completions -H 'Content-Type: application/json' \ + -H 'Authorization: Bearer $AUTH_KEY' \ + -d '{\"model\":\"qwen\",\"messages\":[{\"role\":\"user\",\"content\":\"hi\"}],\"max_tokens\":8}' | grep -iE 'HTTP/|x-routed-peer'" + +# 8.4 monitor 采到该模型(dashboard 或 /api/status) +kubectl -n monitoring exec deploy/monitor -- python3 -c \ + 'import urllib.request,base64,json;r=urllib.request.Request("http://127.0.0.1:8080/api/status");r.add_header("Authorization","Basic "+base64.b64encode(b"admin:").decode());print("api/status",json.load(urllib.request.urlopen(r)) and "OK")' +``` + +`X-Routed-Peer` 头出现 = 请求确实经 cart 路由到了真实后端(cart 的 `proxy.add_routed_peer_header=true`)。 + +--- + +## 9. 升级 / 回滚 + +发新版流程:改代码 → 各仓打 git tag(CI 发布镜像到 Docker Hub)→ chart 在 modelsphere/helm-charts 发布 → 集群 `helm upgrade`。 + +```bash +helm repo update modelsphere +# 不带 --version = 拉最新 tag;--reuse-values 保留安装时的 fullnameOverride / 密钥 / bodylog.host 等 +helm -n llm-route upgrade autoconfig modelsphere/autoconfig --reuse-values +helm -n llm-route upgrade openresty modelsphere/openresty --reuse-values +helm -n llm-route upgrade cart-qwen modelsphere/cart --reuse-values # 每个 cart- 各升一次 +helm -n monitoring upgrade monitor --reuse-values + +helm -n history # 看修订 +helm -n rollback # 回滚到某修订 +helm -n upgrade --version --reuse-values # 需要锁定/回退到某个历史版本才加 --version +``` + +> chart 内容不变时,`helm upgrade` 只更新 release 元数据、**不重启 pod**(渲染出的 spec 一致),零中断。 +> chart 版本不一定等于组件版本(如 `cart` chart 0.2.x 部署的是 CART 0.6.x);锁版本时用 `helm search repo modelsphere/ --versions` 查 chart 版本。 + +--- + +## 10. 卸载 / 清理 + +```bash +kubectl delete -f config/samples/modelroute-qwen.yaml +helm -n llm-route uninstall openresty cart-qwen autoconfig # opt 同理:再 uninstall cart-opt +helm -n monitoring uninstall monitor +kubectl delete -f config/samples/qwen-backend.yaml # 连带删 qwen ns(opt 同理删 opt-backend.yaml) +kubectl delete crd modelroutes.routing.modelsphere.dev # 如需彻底移除 CRD +``` diff --git a/Dockerfile b/Dockerfile index a9bff1b..5849114 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,24 +1,30 @@ -# autoconfig 镜像 —— 多阶段:golang builder 从国内 goproxy 拉依赖编译 → 打进 python:3.12-alpine。 +# autoconfig: the ModelRoute controller (cmd/). Multi-stage: a Go builder +# compiles the binary, which is copied into a small runtime image. # -# 照 llm-openresty/Dockerfile.bodylog:依赖走【国内 goproxy 镜像】(mirrors.tencent.com/go —— public-buildx -# runner 实测可达,见 llm-openresty CI #416384),不再 vendor。go.sum 入库 + GOSUMDB=off → 仍可复现; -# GOTOOLCHAIN=local 防 go 因 go.mod 版本联网拉新工具链。GOPROXY 是 ARG,可 --build-arg 换 aliyun 等。 +# docker build -t autoconfig:dev . # -# docker build -t harbor.4pd.io/hardcore-tech/autoconfig: . -# docker push harbor.4pd.io/hardcore-tech/autoconfig: -# 打 git tag 自动 build+push,见 .gitlab-ci.yml。 -# 基础镜像与 goproxy 都是 ARG,默认走公网(Docker Hub / proxy.golang.org),开箱即可 build。 -# 内网构建加 --build-arg 指向 harbor 缓存与国内 goproxy(见 .gitlab-ci.yml): -# --build-arg GO_BASE=harbor.4pd.io/library/golang:1.23.3-alpine3.20 -# --build-arg RUNTIME_BASE=harbor.4pd.io/hardcore-tech/python:3.12-alpine -# --build-arg GOPROXY=https://mirrors.tencent.com/go/,direct +# Dependencies are not vendored: go.sum is committed, so `go mod download` is +# reproducible. GOTOOLCHAIN=local keeps go from fetching a newer toolchain +# because of the version in go.mod. +# +# The base images and the Go module proxy are build args. Their defaults are +# public (Docker Hub, the upstream Go proxy), so a fresh clone builds as is. +# Behind a firewall, point them at a mirror: +# --build-arg GO_BASE=/library/golang:1.23.3-alpine3.20 +# --build-arg RUNTIME_BASE=/library/python:3.12-alpine +# --build-arg GOPROXY=,direct +# +# A version tag builds and publishes the image: .github/workflows/release.yml. ARG GO_BASE=golang:1.23.3-alpine3.20 ARG RUNTIME_BASE=python:3.12-alpine ARG GOPROXY=https://proxy.golang.org,direct FROM ${GO_BASE} AS build ARG GOPROXY -ENV GOPROXY=${GOPROXY} GOSUMDB=off GOTOOLCHAIN=local CGO_ENABLED=0 GOOS=linux GOARCH=amd64 +# Set by BuildKit for each platform being built; the defaults apply to the legacy builder. +ARG TARGETOS=linux +ARG TARGETARCH=amd64 +ENV GOPROXY=${GOPROXY} GOSUMDB=off GOTOOLCHAIN=local CGO_ENABLED=0 GOOS=${TARGETOS} GOARCH=${TARGETARCH} WORKDIR /src COPY go.mod go.sum ./ RUN go mod download diff --git a/Dockerfile.hagate b/Dockerfile.hagate index ee358cf..aeb52b3 100644 --- a/Dockerfile.hagate +++ b/Dockerfile.hagate @@ -1,15 +1,32 @@ -# autoconfig-hagate 镜像 —— master-standby 的 leader 选举门控 sidecar(cmd/hagate)。 -# openresty/cart/monitor pod 里当 sidecar:只有持 Lease 的 leader 给自己 pod 打 -active 标签, -# Service selector 带它 → 只有 leader 进 Service endpoints(主备)。多阶段:goproxy 国内镜像拉依赖(无 vendor)。 -# docker build -f Dockerfile.hagate -t harbor.4pd.io/hardcore-tech/autoconfig-hagate: . -# 基础镜像与 goproxy 默认走公网;内网构建用 --build-arg 覆盖(见 .gitlab-ci.yml)。 +# autoconfig-hagate: the master-standby leader-election gate sidecar +# (cmd/hagate). In an openresty/cart/monitor pod, only the sidecar that holds +# the Lease labels its own pod -active; the Service selects on that label, +# so only the leader is in the Service endpoints. +# +# docker build -f Dockerfile.hagate -t autoconfig-hagate:dev . +# +# Dependencies are not vendored: go.sum is committed, so `go mod download` is +# reproducible. GOTOOLCHAIN=local keeps go from fetching a newer toolchain +# because of the version in go.mod. +# +# The base images and the Go module proxy are build args. Their defaults are +# public (Docker Hub, the upstream Go proxy), so a fresh clone builds as is. +# Behind a firewall, point them at a mirror: +# --build-arg GO_BASE=/library/golang:1.23.3-alpine3.20 +# --build-arg RUNTIME_BASE=/library/python:3.12-alpine +# --build-arg GOPROXY=,direct +# +# A version tag builds and publishes the image: .github/workflows/release.yml. ARG GO_BASE=golang:1.23.3-alpine3.20 ARG RUNTIME_BASE=python:3.12-alpine ARG GOPROXY=https://proxy.golang.org,direct FROM ${GO_BASE} AS build ARG GOPROXY -ENV GOPROXY=${GOPROXY} GOSUMDB=off GOTOOLCHAIN=local CGO_ENABLED=0 GOOS=linux GOARCH=amd64 +# Set by BuildKit for each platform being built; the defaults apply to the legacy builder. +ARG TARGETOS=linux +ARG TARGETARCH=amd64 +ENV GOPROXY=${GOPROXY} GOSUMDB=off GOTOOLCHAIN=local CGO_ENABLED=0 GOOS=${TARGETOS} GOARCH=${TARGETARCH} WORKDIR /src COPY go.mod go.sum ./ RUN go mod download diff --git a/Dockerfile.reload b/Dockerfile.reload index 9316169..a1ecff2 100644 --- a/Dockerfile.reload +++ b/Dockerfile.reload @@ -1,15 +1,31 @@ -# autoconfig-reload 镜像 —— reload sidecar 独立小程序(cmd/reload)。 -# 与 controller 分离:openresty / CART pod 里挂它当 sidecar,watch ConfigMap 文件变 → SIGHUP 主进程。 -# 多阶段:goproxy 国内镜像拉依赖(无 vendor),同 Dockerfile。GOPROXY 是 ARG,可 --build-arg 换。 -# docker build -f Dockerfile.reload -t harbor.4pd.io/hardcore-tech/autoconfig-reload: . -# 基础镜像与 goproxy 默认走公网;内网构建用 --build-arg 覆盖(见 .gitlab-ci.yml)。 +# autoconfig-reload: the config reload sidecar (cmd/reload). It runs in the +# openresty / CART pod, watches the mounted ConfigMap and sends SIGHUP to the +# main process when a file changes. +# +# docker build -f Dockerfile.reload -t autoconfig-reload:dev . +# +# Dependencies are not vendored: go.sum is committed, so `go mod download` is +# reproducible. GOTOOLCHAIN=local keeps go from fetching a newer toolchain +# because of the version in go.mod. +# +# The base images and the Go module proxy are build args. Their defaults are +# public (Docker Hub, the upstream Go proxy), so a fresh clone builds as is. +# Behind a firewall, point them at a mirror: +# --build-arg GO_BASE=/library/golang:1.23.3-alpine3.20 +# --build-arg RUNTIME_BASE=/library/python:3.12-alpine +# --build-arg GOPROXY=,direct +# +# A version tag builds and publishes the image: .github/workflows/release.yml. ARG GO_BASE=golang:1.23.3-alpine3.20 ARG RUNTIME_BASE=python:3.12-alpine ARG GOPROXY=https://proxy.golang.org,direct FROM ${GO_BASE} AS build ARG GOPROXY -ENV GOPROXY=${GOPROXY} GOSUMDB=off GOTOOLCHAIN=local CGO_ENABLED=0 GOOS=linux GOARCH=amd64 +# Set by BuildKit for each platform being built; the defaults apply to the legacy builder. +ARG TARGETOS=linux +ARG TARGETARCH=amd64 +ENV GOPROXY=${GOPROXY} GOSUMDB=off GOTOOLCHAIN=local CGO_ENABLED=0 GOOS=${TARGETOS} GOARCH=${TARGETARCH} WORKDIR /src COPY go.mod go.sum ./ RUN go mod download diff --git a/Makefile b/Makefile index cb212c1..ce69f3e 100644 --- a/Makefile +++ b/Makefile @@ -1,15 +1,13 @@ # autoconfig —— kubebuilder/operator-sdk 风格 Makefile。 # 工具用 `go run ...@version`(无需装二进制;仅开发时联网拉,不进 CI/镜像)。生成物提交进 repo,CI 只编译。 -IMG ?= harbor.4pd.io/hardcore-tech/autoconfig:latest -RELOAD_IMG ?= harbor.4pd.io/hardcore-tech/autoconfig-reload:latest +IMG ?= 4pdosc/autoconfig:latest +RELOAD_IMG ?= 4pdosc/autoconfig-reload:latest -# Dockerfile 里 GO_BASE / RUNTIME_BASE / GOPROXY 三个 ARG 的默认值是【公网】 -# (Docker Hub + proxy.golang.org),保证外部用户开箱可 build。内网构建覆盖成 harbor -# 缓存与国内 goproxy —— 下面是内网默认值,走公网时 `make docker-build BUILD_ARGS=`。 -BUILD_ARGS ?= --build-arg GO_BASE=harbor.4pd.io/library/golang:1.23.3-alpine3.20 \ - --build-arg RUNTIME_BASE=harbor.4pd.io/hardcore-tech/python:3.12-alpine \ - --build-arg GOPROXY=https://mirrors.tencent.com/go/,direct +# The Dockerfiles default GO_BASE / RUNTIME_BASE / GOPROXY to public sources +# (Docker Hub, the upstream Go proxy). Behind a firewall, override them, e.g. +# make docker-build BUILD_ARGS="--build-arg GOPROXY=,direct" +BUILD_ARGS ?= CONTROLLER_GEN_VERSION ?= v0.16.4 KUSTOMIZE_VERSION ?= v5.4.3 @@ -51,7 +49,7 @@ vet: ; go vet ./... test: manifests generate fmt vet ## 生成 + 静态检查 + 单测。 go test ./... -##@ 构建(module 模式,依赖走 goproxy;与 CI 的 docker build 同源)。本机 goproxy 不通时设 GOPROXY=https://mirrors.tencent.com/go/,direct +##@ 构建(module 模式,依赖走 goproxy;与 CI 的 docker build 同源)。本机 goproxy 不通时设 GOPROXY=,direct .PHONY: build build: ## 交叉编译两个二进制到 bin/。 diff --git a/NOTICE b/NOTICE new file mode 100644 index 0000000..3760721 --- /dev/null +++ b/NOTICE @@ -0,0 +1,23 @@ +autoconfig +Copyright (c) 2026 the ModelSphere authors + +Licensed under the Apache License, Version 2.0. See LICENSE. + +This product includes software developed by third parties. Their licenses apply +to their own code, not to this project's. + +Go modules compiled into the images (direct dependencies, from go.mod) + Kubernetes api, apimachinery, + client-go ......................... Apache-2.0 + controller-runtime ................ Apache-2.0 + fsnotify .......................... BSD-3-Clause + gopkg.in/yaml.v3 .................. MIT / Apache-2.0 + Their own dependencies, listed in go.sum, are under their own licenses. + +Container base images + Runtime: python:3.12-alpine (Docker Hub) — see the image's own licenses. + Build stage only, not shipped: golang:1.23.3-alpine3.20. + +The components autoconfig configures (openresty, the cache-aware router, the +monitor) are separate projects under their own licenses; nothing here +redistributes their code. diff --git a/README.md b/README.md index 535237d..3b7fbe4 100644 --- a/README.md +++ b/README.md @@ -1,220 +1,260 @@ # autoconfig -**让路由层的后端列表跟着 Kubernetes 自动收敛的 operator。** +English | [简体中文](README.zh-CN.md) -在 k8s 上跑推理服务时,后端 pod 的 IP 会随扩缩容、重启、滚动更新不断变化。而前面的路由组件 -—— openresty、cache-aware-router(CART)、监控 —— 各自维护着一份 peer / worker / 采集目标列表。 -人工同步这几份列表既繁琐又容易漏:扩容了没加进去等于白扩,缩容了没摘掉就是持续打死 IP。 +**A Kubernetes operator that keeps the routing layer's backend lists in step with what is actually running.** -autoconfig 用一个 `ModelRoute` 自定义资源描述「一个模型的路由长什么样」,然后: +When you serve inference on Kubernetes, backend pod IPs change all the time: scaling, restarts, rolling +updates. The routing components in front of them — openresty, the cache-aware router (CART), monitoring — +each keep their own list of peers / workers / scrape targets. Syncing those lists by hand is tedious and +easy to get wrong: a scale-up that never reaches the list is wasted, and a scale-down that is never removed +keeps sending traffic to a dead IP. -1. **发现** —— watch 该模型对应 Service 的 EndpointSlice(或 pod label),得到当前就绪的后端端点; -2. **渲染** —— 生成 openresty 的路由配置、CART 的 workers 列表、监控的采集行; -3. **下发** —— 写进各消费方已有的 ConfigMap,由它们各自的 sidecar 热重载生效。 +autoconfig describes "what routing for one model looks like" with a `ModelRoute` custom resource, and then: + +1. **Discovers** — watches the EndpointSlices of the model's Service (or a pod label selector) to get the + backend endpoints that are ready right now; +2. **Renders** — generates the openresty route config, the CART worker list and the monitoring targets; +3. **Delivers** — writes them into ConfigMaps the consumers already have, where each consumer's sidecar + hot-reloads them. ```bash kubectl apply -f config/samples/modelroute-glm.yaml kubectl get mr -A # NAME BACKENDS CART READY AGE ``` -后端扩缩容时不需要任何人工操作,`BACKENDS` 列会自己变。 +When backends scale, nobody has to do anything: the `BACKENDS` column changes on its own. -## 它不做什么 +## What it does not do -- **不创建 chart、不创建 ConfigMap** —— 只把内容写进消费方**已有**的 ConfigMap。消费方的部署、 - 初始配置、sidecar 挂载由各自的 chart 负责(见「消费方接入」)。 -- **不代理流量** —— 它是控制面,数据面仍是 openresty / CART。 -- **不管非 LLM 之外的健康语义** —— `modelType: video` 走纯反向代理,不套 token 级限流那一套。 +- **It creates no charts and no ConfigMaps** — it only writes content into ConfigMaps the consumers + **already have**. Deploying a consumer, its initial config and its sidecar mounts is that consumer's + chart's job (see "Integrating consumers"). +- **It does not proxy traffic** — it is a control plane; openresty and CART remain the data plane. +- **It applies no LLM health semantics to other workloads** — `modelType: video` is rendered as a plain + reverse proxy, without the token-level rate limiting. -## 快速上手 +## Quick start ```bash -# 1. 部署 controller(chart 含 CRD + RBAC + Deployment) +# 1. Deploy the controller (the chart contains the CRD, RBAC and the Deployment) helm upgrade --install autoconfig deploy/helm/autoconfig -n llm-route --create-namespace -# 2. 声明一条路由 +# 2. Declare a route kubectl apply -f config/samples/modelroute-glm.yaml -# 3. 看发现结果 +# 3. Look at what was discovered kubectl get mr -A -kubectl describe mr # status 里有 backends / cartPeers / conditions +kubectl describe mr # status has backends / cartPeers / conditions ``` -不用 helm 时走 kustomize(同源生成物):`make install`(装 CRD)+ `make deploy`(起 controller)。 - -**⚠️ 卸载顺序:先删 ModelRoute,再 `helm uninstall`。** ModelRoute 带 finalizer -(`routing.modelsphere.dev/cleanup`),要 controller 在跑才能摘。若先 uninstall(删了 controller) -再删 ModelRoute / namespace,ModelRoute 会卡住、拖住 namespace 与 CRD 的删除。 -正确顺序:`kubectl delete mr --all -A` → `helm uninstall`。 -(chart 里用 `modelRoutes` 声明的 ModelRoute 由 helm 托管,uninstall 前会随 release 删除、 -controller 还在 → 自动摘 finalizer,无此问题。) -已卡住的补救:`kubectl patch mr -n --type=merge -p '{"metadata":{"finalizers":[]}}'`。 - -## 工作原理 - -![autoconfig 架构:controller 按 Service 发现后端端点 → 写 openresty / CART / monitor 三个 ConfigMap;openresty/CART 里 reload sidecar 收 SIGHUP 热重载,monitor 自身每 60s 热加载](docs/architecture.png) - -三个独立二进制 / 镜像,各司其职: - -| 组件 | 镜像 | 角色 | -|---|---|---| -| **controller** | `autoconfig`(`cmd/`) | 唯一发现逻辑 + RBAC 一处;watch ModelRoute + EndpointSlice + Pod → 发现 → 渲染 → 写 ConfigMap + status。controller 自身 `replicas>1` 时靠 manager 的 leader 选举保证只有一个在干活。 | -| **reload sidecar** | `autoconfig-reload`(`cmd/reload`) | 跑在消费方 pod 里,watch 挂载的 ConfigMap 文件,变化就 `kill -HUP` 主进程(靠 `shareProcessNamespace`)。CART / openresty 收 SIGHUP 优雅重载。 | -| **hagate sidecar** | `autoconfig-hagate`(`cmd/hagate`) | 消费方 **master-standby**:2 副本都保持 Ready,但只有持 Lease 的 leader 给自己 pod 打 `-active=true` 标签;Service selector 带这个标签 → **只有 leader 进 endpoints**。用标签而非 readiness 门控,standby 不会永久 NotReady 卡住滚动。 | - -三个镜像的 tag 与 chart 版本同线(chart 的 `appVersion` = 镜像 tag),见「构建」。 - -## 两个 sidecar 的实现原理 - -消费方 pod(openresty / CART)里除主容器外各挂两个 autoconfig sidecar:**hagate**(主备门控)+ **reload**(配置热重载)。 -两者都靠 `shareProcessNamespace: true` 与主容器同 pod 协作。 - -### hagate —— master-standby 单活门控 +The chart is also published as `modelsphere/autoconfig` in the +[modelsphere Helm repository](https://github.com/modelsphere/helm-charts); installing the whole stack +(autoconfig, CART, openresty, a test model) is described in [DEPLOY.md](DEPLOY.md). -**为什么要单活**:openresty / CART 是**有状态**路由器(openresty 有 session 亲和 + `active_conns` 并发计数, -CART 有 prefix-cache radix tree)。多副本同时进 Service endpoints = 缓存被打散、并发计数分裂,路由质量下降。 -所以要 **2 副本主备(master-standby)**:都保持运行,但同一时刻只有一个对外收流量。 +Without Helm, use kustomize (same generated manifests): `make install` (CRD) + `make deploy` (controller). -**为什么不用 readinessProbe 门控**:若让 standby 的 readiness 恒 NotReady 来挡流量,Deployment 滚动时 -`maxUnavailable`/`minReady` 会把「永久 NotReady 的 standby」当成不可用 → 滚动卡死。故改用**标签门控**而非 readiness。 +**⚠️ Uninstall order: delete the ModelRoutes first, then `helm uninstall`.** A ModelRoute carries a +finalizer (`routing.modelsphere.dev/cleanup`) that only a running controller can remove. If you uninstall +first (removing the controller) and then delete ModelRoutes or their namespace, the ModelRoutes hang and +block the deletion of the namespace and the CRD. The right order: `kubectl delete mr --all -A` → +`helm uninstall`. (ModelRoutes declared through the chart's `modelRoutes` value are managed by Helm: they are +deleted with the release while the controller is still running, so the finalizer is removed and this does +not happen.) To free ModelRoutes that are already stuck: +`kubectl patch mr -n --type=merge -p '{"metadata":{"finalizers":[]}}'`. -**机制**:每 pod 一个 hagate sidecar 参与 Lease `-ha` 的 leader 选举。 -- 只有持 Lease 的 leader 给**自己 pod** 打 `-active=true` 标签; -- Service 的 selector 带这个标签 → **只有 leader 的 pod 进 endpoints**,standby 在池外待命; -- 消费方上游(如 openresty 的 cart 层)走 **Service ClusterIP VIP**,VIP 恒指 active leader → failover/rollout 对上游透明。 +## How it works -**level-triggered 自愈**:每 2s 读 pod 实际标签,与「**该不该 active**(= 是 leader **且**本地 app 端口可连)」比对, -不符就纠正 —— 标签被外部误删也能自愈;本地 app 端口连不上时即便是 leader 也主动摘标签(避免把流量导向坏 pod)。 +![autoconfig architecture: the controller discovers backend endpoints per Service → writes the openresty / CART / monitor ConfigMaps; the reload sidecar in openresty/CART hot-reloads on SIGHUP, the monitor reloads itself every 60s](docs/architecture.png) -**failover**:计划内下线(SIGTERM=删 pod/滚动/驱逐)release Lease → standby ~1-2s 接管**新**流量; -本 pod **保留** active 标签作 **terminating endpoint**(deletionTimestamp),靠 CNI 的 graceful-terminating -把新连接导向 standby、老在途连接留在本 pod 排空(配合主容器优雅停 + grace),**不硬摘标签 → 不 reset 在途连接**, -pod 退出即自动出 endpoints。存活丢主(Lease 续约失败但 pod 没死)才摘标签离开 Service(避免 2-active)。 -硬崩则等 Lease TTL 过期后接管。 +Three separate binaries / images, each with one job: -**监控组件不做 HA**:采集/告警是**自主轮询循环**(不接收外部流量),readiness 门控挡不住重复采集, -单活无意义 → 单例(`replicas: 1` + `Recreate`),chart 不带 hagate。 -openresty/cart 默认开(`replicas: 2` + `ha.enabled: true`)。 - -### reload —— 配置热重载 - -**问题**:autoconfig 改写了 ConfigMap,主进程(nginx / CART)要重读配置才生效,但不能重启(会断在途长流式连接)。 - -**机制**:每消费方 pod 一个 reload sidecar: -- 把输出 ConfigMap **整卷挂**(非 subPath —— subPath 不随 ConfigMap 更新同步)到 `--watch` 目录,用 fsnotify 监听; -- 文件变 → 找主进程 pid(读 `/proc/*/cmdline` 匹配 `nginx: master` / `cache-aware-router`; - **用 cmdline 不用 `comm`** —— comm 截断 15 字符、且要避开 nginx worker)→ `kill -HUP`; -- nginx / CART 收 **SIGHUP 都是优雅重载**:坏配置只 log warning + 保留旧配置,绝不中断在途请求。 - -**传播延迟**:kubelet 同步挂载的 ConfigMap 有 ~1min 延迟(AtomicWriter `..data` 原子软链切换 → -reload 看到的永远是完整文件,不会读到半写)。 - -**openresty 侧 `--sock-dir`**:per-model server 监听 unix socket,模型删除后 nginx 不会自动 unlink -残留 `.sock`;reload 前先删掉「已无 conf 引用」的孤儿 socket。 - -## ModelRoute CRD - -`ModelRoute`(`routing.modelsphere.dev/v1alpha1`),一个模型一个对象,`kubectl apply` 当场 CEL 校验、 -`kubectl get mr` 看发现结果。完整样例见 [`config/samples/modelroute-glm.yaml`](config/samples/modelroute-glm.yaml)。 -下表逐字段说明(✅=必填)。 - -**`spec` 顶层** - -| 字段 | 必填 | 含义 | +| Component | Image | Role | +|---|---|---| +| **controller** | `autoconfig` (`cmd/`) | The only discovery logic and the only place with RBAC. Watches ModelRoute + EndpointSlice + Pod → discovers → renders → writes ConfigMaps + status. With `replicas>1`, the manager's leader election makes sure only one replica does the work. | +| **reload sidecar** | `autoconfig-reload` (`cmd/reload`) | Runs in the consumer's pod, watches the mounted ConfigMap files and `kill -HUP`s the main process when they change (needs `shareProcessNamespace`). CART and openresty reload gracefully on SIGHUP. | +| **hagate sidecar** | `autoconfig-hagate` (`cmd/hagate`) | Consumer **master-standby**: both replicas stay Ready, but only the leader holding the Lease labels its own pod `-active=true`; the Service selects on that label → **only the leader is in the endpoints**. Gating with a label instead of readiness means the standby is never permanently NotReady and never stalls a rollout. | + +The three images share one version number with each other and with the chart (`appVersion` = image tag); +see "Building". + +## How the two sidecars work + +Besides the main container, each consumer pod (openresty / CART) carries two autoconfig sidecars: +**hagate** (master-standby gate) and **reload** (config hot reload). Both work alongside the main container +through `shareProcessNamespace: true`. + +### hagate — single-active master-standby gate + +**Why single-active**: openresty and CART are **stateful** routers (openresty keeps session affinity and an +`active_conns` concurrency count; CART keeps a prefix-cache radix tree). With several replicas in the +Service endpoints at once, the cache is scattered and the concurrency count split, and routing quality +drops. So they run **two replicas as master-standby**: both running, only one taking traffic at a time. + +**Why not gate with a readinessProbe**: if the standby were kept NotReady to keep traffic away, a Deployment +rollout would count that permanently NotReady standby as unavailable under `maxUnavailable`/`minReady` and +stall. Hence a **label gate** instead of readiness. + +**Mechanism**: one hagate sidecar per pod takes part in leader election on the Lease `-ha`. +- Only the leader holding the Lease labels **its own pod** `-active=true`; +- The Service's selector includes that label → **only the leader's pod is in the endpoints**; the standby + waits outside the pool; +- Consumers upstream (for example openresty's cart tier) go through the **Service ClusterIP VIP**, which + always points at the active leader → failover and rollouts are transparent to them. + +**Level-triggered self-healing**: every 2s it reads the pod's actual label, compares it with whether the pod +**should be active** (= it is the leader **and** the local app port accepts connections), and corrects any +difference — so a label removed by someone else heals itself, and a leader whose local app port is unreachable +removes its own label rather than send traffic to a broken pod. + +**Failover**: on a planned shutdown (SIGTERM: pod deletion, rollout, eviction) it releases the Lease → the +standby takes over **new** traffic in about 1–2s. The old pod **keeps** its active label and stays a +**terminating endpoint** (deletionTimestamp set); the CNI's graceful-termination handling sends new +connections to the standby while in-flight connections drain on the old pod (together with the main +container's graceful stop and grace period). **The label is not torn off, so in-flight connections are not +reset**; the pod leaves the endpoints when it exits. Only when leadership is lost while the pod is alive +(Lease renewal fails) does it remove the label and leave the Service, to avoid two active pods. After a hard +crash, the standby takes over when the Lease TTL expires. + +**The monitoring component has no HA**: scraping and alerting are a **self-driven polling loop** (it takes no +external traffic), so a readiness gate cannot stop duplicate scraping and single-active buys nothing → it +runs as a singleton (`replicas: 1` + `Recreate`), and its chart has no hagate. +openresty and CART enable it by default (`replicas: 2` + `ha.enabled: true`). + +### reload — config hot reload + +**Problem**: once autoconfig rewrites a ConfigMap, the main process (nginx / CART) must re-read its config, +but must not restart (that would cut long in-flight streaming connections). + +**Mechanism**: one reload sidecar per consumer pod: +- The output ConfigMap is **mounted as a whole volume** (not with subPath — a subPath mount does not follow + ConfigMap updates) at the `--watch` directory and watched with fsnotify; +- When a file changes → find the main process's pid (read `/proc/*/cmdline` and match `nginx: master` / + `cache-aware-router`; **cmdline rather than `comm`**, since comm is truncated to 15 characters, and nginx + workers must not match) → `kill -HUP`; +- **SIGHUP is a graceful reload for both nginx and CART**: a bad config only logs a warning and keeps the old + one, and in-flight requests are never interrupted. + +**Propagation delay**: the kubelet syncs mounted ConfigMaps with a delay of about a minute (the AtomicWriter +swaps the `..data` symlink atomically, so reload always sees complete files, never a half-written one). + +**`--sock-dir` for openresty**: per-model servers listen on unix sockets, and nginx does not unlink a model's +leftover `.sock` after the model is deleted; before reloading, the sidecar deletes orphan sockets that no conf +references any more. + +## The ModelRoute CRD + +`ModelRoute` (`routing.modelsphere.dev/v1alpha1`), one object per model. `kubectl apply` validates it on the +spot with CEL, and `kubectl get mr` shows what was discovered. A full example is +[`config/samples/modelroute-glm.yaml`](config/samples/modelroute-glm.yaml). The tables below describe each +field (✅ = required). + +**Top-level `spec`** + +| Field | Required | Meaning | |---|---|---| -| `modelType` | 可选 | 这条路由服务的是哪类模型,决定渲染方式。默认 `llm`;`video` = 视频生成,见下 | -| `discovery` | ✅ | 本模型的后端桶发现方式(喂 CART workers / nginx backend / monitor services) | -| `cart` | 可选 | 配了 = autoconfig 管这个 CART;省略 = nginx 直连后端(无 CART 层) | -| `nginx` | ✅ | openresty 路由:渲染 peers → `session_route_.conf` | -| `monitor` | 可选 | 把发现的后端/入口/CART 也写进共享的监控配置 | +| `modelType` | optional | What kind of model this route serves; decides how it is rendered. Default `llm`; `video` = video generation, see below | +| `discovery` | ✅ | How this model's backend bucket is discovered (feeds CART workers / nginx backends / monitor services) | +| `cart` | optional | Set = autoconfig manages this CART; omitted = nginx talks to the backends directly (no CART tier) | +| `nginx` | ✅ | openresty routing: peers rendered → `session_route_.conf` | +| `monitor` | optional | Also write the discovered backends / entry point / CART into the shared monitoring config | -### modelType:一条路由服务哪类模型 +### modelType: what kind of model a route serves -| 取值 | 渲染成什么 | 适用 | +| Value | Rendered as | For | |---|---|---| -| `llm`(默认) | 走 lua 路由引擎:会话亲和、TTFT/TPS 限流、自适应并发、CART 前置 | `/v1/chat/completions` 这类 token 流式接口 | -| `video` | **纯反向代理**:不解析请求体、不限流;放开超时、关响应缓冲、透传 `Range`、补齐 `X-Forwarded-Host/Proto` | 视频生成:异步建任务 + 轮询 + 大文件下载 | +| `llm` (default) | The Lua routing engine: session affinity, TTFT/TPS limits, adaptive concurrency, CART in front | Token-streaming APIs such as `/v1/chat/completions` | +| `video` | **A plain reverse proxy**: no request-body parsing, no rate limiting; long timeouts, response buffering off, `Range` passed through, `X-Forwarded-Host/Proto` filled in | Video generation: asynchronous job creation + polling + large file downloads | -为什么视频不能套 LLM 那套:请求体可能是 64MB 的 base64 图(引擎要解析 body 取 model)、 -一条片子要 1~3 分钟才出结果(TTFT/TPS 这类 token 级指标无从谈起)、响应是几十 MB 的视频流 -(响应缓冲会把它憋在内存或磁盘上),而下载接口还要支持断点续传(`Range` 必须原样透传)。 +Why video cannot reuse the LLM setup: a request body can be a 64 MB base64 image (the engine would parse the +body to find the model); one clip takes 1–3 minutes (token-level metrics such as TTFT/TPS mean nothing); +the response is a video stream of tens of MB (response buffering would hold it in memory or on disk); and +downloads must support resuming (`Range` has to pass through untouched). -`video` 的可用调优项如下(其余会被 CEL 拒,避免「配了以为生效」): +The tuning keys available for `video` are listed below (CEL rejects any other key, so nothing gets configured +in the belief that it has an effect): -| `nginx.values` 键 | 默认 | 含义 | +| `nginx.values` key | Default | Meaning | |---|---|---| -| `max_body_size` | `64m` | 请求体上限(I2V 允许 base64 传图) | -| `proxy_timeout` | `3600s` | 读/写超时(生成 + 大文件下载) | -| `connect_timeout` | `10s` | 连后端超时 | -| `rate_limit` | **不配 = 不限速** | **单连接**下载限速,配了才渲染 `limit_rate`。接受 `200Mbps`/`1.5Gbps`(比特口径,自动换算成 nginx 要的字节/秒)或 nginx 原生写法(`25m`/`512k`) | -| `rate_limit_after` | `1m`(仅当配了 `rate_limit`) | 前 N 字节全速。建任务/查询/删除都是几百字节的 JSON,不该被下载限速拖慢 | -| `upload_conn_limit` | 不配 = 不限 | 每 IP 同时在传的连接数(`limit_conn`),超出直接 503 | -| `upload_req_limit` | 不配 = 不限 | 每 IP 请求速率(`limit_req`,nginx 原生写法如 `10r/s`) | -| `upload_req_burst` | 不配 = 无突发 | 配合 `upload_req_limit` 的突发额度 | -| `api_keys` | 不配 = **不鉴权** | 逗号分隔的 Bearer token,与 LLM 路由同一套约定;不匹配返回 401 | -| `auth_public_paths` | `~^/v2/video_generation/[^/]+/content$` | 免鉴权的路径(nginx map 左值)。默认放行下载:`content.url` 交给最终用户,浏览器不带 Authorization 头,而任务 id 是 UUID、相当于一次性能力 URL。置空 = 连下载也要 key | -| `upload_limit_key` | `$http_x_real_ip` | 上面两个 zone 按什么分组。**不能用 `$binary_remote_addr`**,原因见文末 | - -**下载限速必须配合开缓冲**:`proxy_buffering off` 时 `limit_rate` 会被 nginx 完全忽略 -(50MB 实测:静态文件 4.99s / 开缓冲 4.61s / 关缓冲 0.089s,`proxy_limit_rate` 同理)。 -所以配了 `rate_limit` 时模板渲染成 `proxy_buffering on` + `proxy_max_temp_file_size 0` -—— 开缓冲但不落临时文件,缓冲区满即对上游反压;不配限速时仍是 `proxy_buffering off` 边收边发。 - -**上传方向没有字节级限速**:`limit_rate`/`proxy_limit_rate` 都只作用于响应,nginx 没有 -限制请求体读取速率的指令(真要做只能在 lua 里自己读 `ngx.req.socket` 加 sleep,会丢掉 -`proxy_request_buffering` 的现成反压)。所以上传靠三道闸:`max_body_size` 卡单条体积、 -`upload_conn_limit` 卡并发、`upload_req_limit` 卡频率 —— 单个来源的入向带宽 ≈ 并发数 × 单条速率。 - -限速只限**速度不限大小** —— `client_max_body_size` 管的是请求体,和响应无关; -`proxy_buffering off` 也让响应不落临时文件,所以下载的视频多大都行(1GB 按 200Mbps 约 40 秒)。 -`proxy_read_timeout` 限的是两次数据之间的间隔,不是总时长。 - -CEL 还会拒掉 `video` + `cart` / `slo` / `monitor`:前两个是 LLM 专用;监控的探活与告警 -按 LLM 端点设计,指向视频服务只会产生假告警(用 Prometheus 抓服务自己的指标)。 - -peers 的优先级在 `video` 下映射成 nginx 的主用/`backup` 两档:优先级最高的一组主用, -更低的(如 `backend-svc` 这种 VIP 静态兜底)标 `backup`,pod-IP 那层全挂了才顶上。 - -**下发通道与 llm 完全一致**:同样写进 `nginx.outputConfigMap` 的 `session_route_.conf` 键, -同样由 reload sidecar 监听挂载目录 → `SIGHUP` 生效,没有第二条通道。 -(sidecar 还靠 conf 里的 `listen unix:.../.sock;` 判断哪些 socket 仍在用, -video 模板保持同样的 listen 行格式,有用例守着。) - -样例见 [`config/samples/modelroute-minimax-h3.yaml`](config/samples/modelroute-minimax-h3.yaml)。 - -**`spec.discovery`** —— 一桶后端怎么发现 - -| 字段 | 类型 | 默认/约束 | 含义与配置 | +| `max_body_size` | `64m` | Request body limit (I2V allows base64 images) | +| `proxy_timeout` | `3600s` | Read/write timeout (generation + large downloads) | +| `connect_timeout` | `10s` | Timeout for connecting to the backend | +| `rate_limit` | **unset = no limit** | **Per-connection** download rate limit; `limit_rate` is rendered only when set. Accepts `200Mbps`/`1.5Gbps` (bits, converted to the bytes/second nginx wants) or nginx's own notation (`25m`/`512k`) | +| `rate_limit_after` | `1m` (only when `rate_limit` is set) | The first N bytes go at full speed. Creating, querying and deleting jobs are a few hundred bytes of JSON and should not be slowed by the download limit | +| `upload_conn_limit` | unset = no limit | Concurrent connections per IP (`limit_conn`); excess requests get 503 | +| `upload_req_limit` | unset = no limit | Request rate per IP (`limit_req`, nginx notation such as `10r/s`) | +| `upload_req_burst` | unset = no burst | Burst allowance for `upload_req_limit` | +| `api_keys` | unset = **no authentication** | Comma-separated Bearer tokens, the same convention as LLM routes; a mismatch returns 401 | +| `auth_public_paths` | `~^/v2/video_generation/[^/]+/content$` | Paths exempt from authentication (left side of an nginx map). Downloads are exempt by default: `content.url` is handed to end users, browsers send no Authorization header, and the job id is a UUID, effectively a one-time capability URL. Empty = downloads need a key too | +| `upload_limit_key` | `$http_x_real_ip` | What the two zones above are keyed by. **`$binary_remote_addr` does not work**; see the end of this document | + +**A download rate limit needs buffering on**: with `proxy_buffering off`, nginx ignores `limit_rate` +entirely (measured on 50 MB: static file 4.99s / buffering on 4.61s / buffering off 0.089s; `proxy_limit_rate` +behaves the same). So when `rate_limit` is set, the template renders `proxy_buffering on` + +`proxy_max_temp_file_size 0` — buffering without temporary files, so a full buffer applies backpressure +upstream. Without a rate limit it stays `proxy_buffering off`, streaming as it receives. + +**There is no byte-level limit on uploads**: `limit_rate`/`proxy_limit_rate` only apply to responses, and +nginx has no directive that limits how fast a request body is read (doing it would mean reading +`ngx.req.socket` in Lua with sleeps, losing the backpressure `proxy_request_buffering` already provides). So +uploads are bounded by three gates: `max_body_size` caps the size of one request, `upload_conn_limit` caps +concurrency, and `upload_req_limit` caps frequency — one source's inbound bandwidth ≈ concurrency × per-request +rate. + +A rate limit limits **speed, not size** — `client_max_body_size` applies to the request body, not the +response, and `proxy_buffering off` keeps responses out of temporary files, so downloaded videos can be any size +(1 GB at 200 Mbps takes about 40 seconds). `proxy_read_timeout` limits the gap between two reads, not the total +duration. + +CEL also rejects `video` combined with `cart` / `slo` / `monitor`: the first two are LLM-specific; the +monitor's liveness checks and alerts are designed for LLM endpoints and would only raise false alarms against +a video service (scrape the service's own metrics with Prometheus instead). + +Under `video`, peer priorities map to nginx's two tiers, primary and `backup`: the highest-priority group is +primary, and lower ones (such as the `backend-svc` VIP fallback) are marked `backup` and only take over when the +whole pod-IP tier is down. + +**Delivery is exactly the same as for `llm`**: written to the `session_route_.conf` key of +`nginx.outputConfigMap`, picked up by the reload sidecar watching the mounted directory → `SIGHUP`. There is no +second channel. (The sidecar also uses the conf's `listen unix:.../.sock;` line to tell which sockets are +still in use; the video template keeps the same listen line format, and a test guards it.) + +Example: [`config/samples/modelroute-minimax-h3.yaml`](config/samples/modelroute-minimax-h3.yaml). + +**`spec.discovery`** — how one bucket of backends is discovered + +| Field | Type | Default / constraint | Meaning | |---|---|---|---| -| `service` | string | 与 `selector` **二选一** | EndpointSlice 发现(推荐);支持 `ns/name` 跨 ns(裸名默认同 ModelRoute 的 ns)→ ModelRoute 可放中心 ns | -| `selector` | string | 与 `service` **二选一** | pod label 发现(没建 Service 的单机/单卡兜底) | -| `port` | int | 可选,省略自动推导 | 后端端口;service 路径从 EndpointSlice 取、selector 路径从 containerPort 取(仅单端口可推) | -| `includeNotReady` | bool | `false` | 默认只取 Ready 端点(排空中端点自动排除);`true` = 含未 Ready | +| `service` | string | **exactly one of** `service` / `selector` | EndpointSlice discovery (recommended); `ns/name` works across namespaces (a bare name means the ModelRoute's namespace) → ModelRoutes can live in a central namespace | +| `selector` | string | **exactly one of** `service` / `selector` | Pod label discovery (fallback for single-node / single-GPU backends without a Service) | +| `port` | int | optional, derived when omitted | Backend port; taken from the EndpointSlice for `service`, from the containerPort for `selector` (only a single port can be derived) | +| `includeNotReady` | bool | `false` | By default only Ready endpoints are used (draining endpoints are excluded); `true` = include not-ready ones | -**`spec.cart`** —— 省略整段 = 无 CART +**`spec.cart`** — omit the whole block for no CART -| 字段 | 类型 | 默认/约束 | 含义与配置 | +| Field | Type | Default / constraint | Meaning | |---|---|---|---| -| `service` / `selector` | string | **二选一** | CART pod 发现(供 openresty 的 cart source);`service` 支持 `ns/name` | -| `port` | int | 省略推导 | CART 端口(EndpointSlice/containerPort 单端口自动推) | -| `outputConfigMap` | string | ✅ | 写 CART `config.yaml` 的目标 `ns/name`;底稿(server/cache/health)由 CART chart 的 `values.baseConfig` 建在此 CM,autoconfig 只重填 `workers` 段 | -| `maxLoad` | int | `20` | 每 worker 的 `max_load` | +| `service` / `selector` | string | **exactly one** | How the CART pods are discovered (for openresty's cart source); `service` accepts `ns/name` | +| `port` | int | derived when omitted | CART port (derived from a single EndpointSlice port / containerPort) | +| `outputConfigMap` | string | ✅ | `ns/name` of the ConfigMap CART's `config.yaml` is written to; the base config (server/cache/health) is created in it by the CART chart's `values.baseConfig`, and autoconfig only rewrites the `workers` section | +| `outputKey` | string | `config.yaml` | Which key of that ConfigMap the workers go to. Point it at a workers-only key (for example `workers.yaml`) to leave the base config key entirely to the chart; CART merges `-c config.yaml -c workers.yaml` in that order | +| `maxLoad` | int | `20` | `max_load` per worker | -**`spec.nginx`** —— openresty 路由 +**`spec.nginx`** — openresty routing -| 字段 | 类型 | 默认/约束 | 含义与配置 | +| Field | Type | Default / constraint | Meaning | |---|---|---|---| -| `route` | string | 省略 = `metadata.name` | 路由短名 = conf 文件名 + openresty dict 名 + `.sock` + 外部路径 key `//`。**字符集必须 ⊆ `[a-z0-9._-]`**(做 dispatch 路径捕获正则;含大写会派生不到 socket → 8080 打不通) | -| `peers` | list | ✅(≥1) | 有序 peer 组(见 `peers[]` 表) | -| `outputConfigMap` | string | ✅ | 输出 ConfigMap `ns/name`(多路由共享,每路由一个 key,finalizer 摘各自 key) | -| `values` | map | 可选 | 任意调优项原样渲染进 lua `register_route` 返回表(key=value)→ 加新调优项无需改代码 | -| `service` | string | 可选 | nginx 入口自身的 Service `ns/name` → 供监控的 `nginx:` 行 + 入口 pod 扩缩事件驱动;端口取 Service 的 dispatch 命名端口 8080 | -| `selector` | string | 可选 | nginx 入口 pod label 发现(没建 Service 兜底;与 `service` 二选一,都配则 `service` 优先) | +| `route` | string | omitted = `metadata.name` | Short route name = conf file name + openresty dict name + `.sock` + external path key `//`. **Characters must be within `[a-z0-9._-]`** (it is captured by the dispatch path regex; with upper case no socket is derived and port 8080 cannot reach it) | +| `peers` | list | ✅ (≥1) | Ordered peer groups (see the `peers[]` table) | +| `outputConfigMap` | string | ✅ | Output ConfigMap `ns/name` (shared by many routes, one key per route; the finalizer removes each route's own key) | +| `values` | map | optional | Any tuning keys, rendered as-is into the table returned by Lua `register_route` (key=value) → a new tuning key needs no code change | +| `service` | string | optional | The nginx entry point's own Service `ns/name` → used for the monitor's `nginx:` lines and to react to entry pods scaling; the port is the Service's named dispatch port 8080 | +| `selector` | string | optional | nginx entry pods by label (fallback without a Service; alternative to `service`, which wins if both are set) | -**`spec.nginx.values` 示例 —— 开启动态限流(自适应并发 AIMD)** +**`spec.nginx.values` example — dynamic rate limiting (adaptive concurrency, AIMD)** -配了 `tps_limit_tps` 即对本路由 opt-in;未显式 `adaptive_cc: "false"` 时,`adaptive_cc` 按全局默认自动开。 -删掉 `values` 即退回不限流。 +Setting `tps_limit_tps` opts the route in; unless `adaptive_cc: "false"` is set explicitly, `adaptive_cc` +follows the global default and turns on. Removing `values` returns to no rate limiting. ```yaml spec: @@ -222,97 +262,110 @@ spec: route: qwen service: llm-route/openresty outputConfigMap: llm-route/openresty-conf - values: # 任意 key 原样渲染进 lua register_route opts(值必须字符串) - tps_limit_tps: "30" # 解码速率下限(tok/s):EWMA 低于它→AIMD 缩并发,高于它→涨(= opt-in 闸门) - # adaptive_cc_min: "10" # 可选:并发下限(不配 = 静态 max × 全局 min_frac 派生) - # ttft_limit_ms: "60000" # 可选:TTFT 软控阈值 - # adaptive_cc: "false" # 可选:显式关自适应,走静态硬熔断 + values: # any key, rendered as-is into the Lua register_route opts (values must be strings) + tps_limit_tps: "30" # decode-rate floor (tok/s): EWMA below it → AIMD lowers concurrency, above it → raises (= the opt-in switch) + # adaptive_cc_min: "10" # optional: concurrency floor (unset = static max × global min_frac) + # ttft_limit_ms: "60000" # optional: soft TTFT threshold + # adaptive_cc: "false" # optional: turn adaptive concurrency off and use the static hard limit peers: - { use: cart, priority: 3, maxConcurrencyFromBackend: true } - { use: backend, priority: 2, maxConcurrency: 100 } - { use: backend-svc, priority: 1 } ``` -生效后 openresty 侧 `GET //_tps_status` 应见 `opt_in=true, adaptive_cc_on=true`。 +Once applied, `GET //_tps_status` on openresty should show `opt_in=true, adaptive_cc_on=true`. -**`spec.nginx.peers[]`** —— 有序分层(数字大=优先,高优层全 banned 才级联到低层) +**`spec.nginx.peers[]`** — ordered tiers (higher number = preferred; traffic falls through to a lower tier +only when every peer of the higher tier is banned) -| 字段 | 类型 | 默认/约束 | 含义与配置 | +| Field | Type | Default / constraint | Meaning | |---|---|---|---| -| `use` | enum | ✅ `cart`\|`backend`\|`backend-svc` | 见下「三档 `use`」 | -| `priority` | int | — | openresty peer 优先级(建议 cart=3、backend=2、backend-svc=1) | -| `maxConcurrency` | int | 省略用 `values.default_max` | 该组所有 peer 的并发上限 | -| `maxConcurrencyFromBackend` | bool | 仅 `use:cart` 有意义 | `true` = cart 并发上限动态 = 后端单实例并发 × 后端数(CART 扇出到 N 后端,容量随扩缩自动变);设了则忽略静态 `maxConcurrency`,且**要求 `backend` 组 `maxConcurrency>0`** 作乘数 | -| `probePath` | string | 省略见右 | openresty 健康探测路径覆盖(GET,状态行含 200=健康否则 ban)。`use:cart` 默认 `/health`(CART 的 `/v1/models` 是缓存端点、worker 全挂也返 200,不能当信号);其余层默认空 = 用 route 的 `health_probe_path` | - -三档 `use`: -- **`cart`**(priority 3)—— CART 上游,走 **CART Service 的 ClusterIP(VIP,非 pod IP)**:CART 是 master-standby, - VIP 恒指 active leader → CART failover/rollout 对 openresty 透明,autoconfig 无需重写。 -- **`backend`**(priority 2)—— 后端 **pod IP**(session 亲和 / least_conn / per-peer 健康的主力层)。 -- **`backend-svc`**(priority 1,可选兜底)—— 后端 **Service 的 ClusterIP(VIP)静态兜底**: - **autoconfig 本身宕 + 后端 rollout** 时 pod-IP 层是死 IP 又没人重写 → 若无兜底会全断; - VIP 由 kube-proxy 维护、不依赖 autoconfig 存活,pod-IP 层全 banned 后级联到它 → - **降级(走 kube-proxy、无亲和)但不全断**。需 `discovery.service`(selector 模式无 VIP,自动跳过)。 - -**`spec.monitor`** —— 可选;监控组件自身每 60s 热加载,**无 reload sidecar**(区别于 nginx/CART)。每模型一个 key,三类行: - -| 字段 | 类型 | 默认/约束 | 含义与配置 | +| `use` | enum | ✅ `cart`\|`backend`\|`backend-svc` | See "The three `use` values" below | +| `priority` | int | — | openresty peer priority (suggested: cart=3, backend=2, backend-svc=1) | +| `maxConcurrency` | int | omitted = `values.default_max` | Concurrency limit for every peer in the group | +| `maxConcurrencyFromBackend` | bool | only meaningful for `use:cart` | `true` = the cart limit is dynamic = per-backend concurrency × number of backends (CART fans out to N backends, so capacity follows scaling); overrides a static `maxConcurrency`, and **requires `maxConcurrency>0` on the `backend` group** as the multiplier | +| `probePath` | string | see right when omitted | Overrides openresty's health probe path (GET; a status line containing 200 = healthy, otherwise banned). `use:cart` defaults to `/health` (CART's `/v1/models` is cached and returns 200 even when every worker is down, so it is no signal); other tiers default to empty = the route's `health_probe_path` | + +The three `use` values: +- **`cart`** (priority 3) — the CART upstream, through the **CART Service's ClusterIP (VIP, not pod IPs)**: CART + runs master-standby and the VIP always points at the active leader → CART failover and rollouts are + transparent to openresty, and autoconfig need not rewrite anything. +- **`backend`** (priority 2) — backend **pod IPs** (the main tier for session affinity / least_conn / + per-peer health). +- **`backend-svc`** (priority 1, optional fallback) — the backend **Service's ClusterIP (VIP) as a static + fallback**: if **autoconfig itself is down while the backend rolls out**, the pod-IP tier holds dead IPs and + nobody rewrites it → without a fallback everything fails. The VIP is maintained by kube-proxy and does not + depend on autoconfig; when the whole pod-IP tier is banned, traffic falls through to it → **degraded (through + kube-proxy, no affinity) but not down**. Needs `discovery.service` (selector mode has no VIP and skips it). + +**`spec.monitor`** — optional. The monitoring component reloads itself every 60s, **with no reload sidecar** +(unlike nginx/CART). One key per model, three kinds of lines: + +| Field | Type | Default / constraint | Meaning | |---|---|---|---| -| `outputConfigMap` | string | ✅ | 写监控配置的 ConfigMap `ns/name`(多模型共享,每模型一个 key) | -| `model` | string | 省略 = `metadata.name` | `service:` 行的 model 字段(served-model-name) | -| `gpuType` | string | 省略自动推导 | `service:` 行的 gpu_type;省略 = 从后端节点 GPU label `nvidia.com/gpu.product`(GFD)推短名,推不出留空 | -| `nginx` | bool | 默认 `true`(当 `spec.nginx` 配了 service/selector) | 复用 nginx 入口发现写 `nginx:` 行;`false` 关 | -| `router` | bool | 默认 `true`(当配了 `spec.cart`) | 复用 `spec.cart` 发现的 CART pod 写 `router:` 表(`.../workers`);`false` 关 | +| `outputConfigMap` | string | ✅ | ConfigMap `ns/name` the monitoring config is written to (shared by many models, one key per model) | +| `model` | string | omitted = `metadata.name` | The model field of the `service:` lines (served-model-name) | +| `gpuType` | string | derived when omitted | The gpu_type of the `service:` lines; omitted = a short name derived from the backend node's GPU label `nvidia.com/gpu.product` (GFD), left empty if it cannot be derived | +| `nginx` | bool | default `true` (when `spec.nginx` has a service/selector) | Reuse the nginx entry discovery to write `nginx:` lines; `false` turns it off | +| `router` | bool | default `true` (when `spec.cart` is set) | Reuse the CART pods discovered for `spec.cart` to write the `router:` lines (`.../workers`); `false` turns it off | -三类输出行格式:`service: \| \| \| `(每后端实例一行)、 -`nginx: - \| http://ip:8080/`、`router: -router- \| http://ip:port/workers`。 +The three output line formats: `service: \| \| \| ` (one line per backend +instance), `nginx: - \| http://ip:8080/`, `router: -router- \| http://ip:port/workers`. -**CEL 校验(apply 时即报错)**:① `discovery`/`cart` 的 `service` 与 `selector` 必须**恰好一个**; -② `nginx.peers` 用了 `cart` 必须配 `spec.cart`;③ `cart` 用 `maxConcurrencyFromBackend` 必须给 -`backend` 组配 `maxConcurrency>0`。 +**CEL validation (errors at apply time)**: ① for `discovery`/`cart`, **exactly one** of `service` and +`selector`; ② if `nginx.peers` uses `cart`, `spec.cart` must be set; ③ if `cart` uses +`maxConcurrencyFromBackend`, the `backend` group must have `maxConcurrency>0`. -## 消费方接入 +## Integrating consumers -autoconfig 只负责把配置**写进已有的 ConfigMap**(不创建 chart、不创建 ConfigMap);消费方各自 chart 建初始 -ConfigMap + reload/hagate sidecar + Service 门控,并把对应 ConfigMap 挂进自己的 pod。 -`autoconfig-reload` / `autoconfig-hagate` 两个 sidecar 镜像由本仓构建,消费方跨仓引用。三个消费方一览: +autoconfig only **writes config into existing ConfigMaps** (it creates no charts and no ConfigMaps). Each +consumer's chart creates the initial ConfigMap, the reload/hagate sidecars and the Service gate, and mounts the +ConfigMap into its pod. The `autoconfig-reload` / `autoconfig-hagate` sidecar images are built from this +repository and referenced by the consumers. The three consumers: -| 消费方 | autoconfig 写的 ConfigMap → 挂载文件 | reload sidecar | +| Consumer | ConfigMap autoconfig writes → mounted file | reload sidecar | |---|---|---| | **openresty** | `openresty-conf` → `conf.d/routes/session_route_.conf` | ✅ `--process "nginx: master"` + `--sock-dir` | -| **cart**(cache-aware-router) | `cart-config` → `configs/config.yaml`(只重填 `workers` 段) | ✅ `--process cache-aware-router` | -| **监控** | `monitor-conf` → `conf.d/.monitor.conf` | ❌ 自身每 60s 热加载 | +| **cart** (cache-aware-router) | `cart-config` → `configs/config.yaml` (only the `workers` section is rewritten) | ✅ `--process cache-aware-router` | +| **monitoring** | `monitor-conf` → `conf.d/.monitor.conf` | ❌ reloads itself every 60s | ### openresty -- **ConfigMap 交付(为什么挂子目录)**:ConfigMap 整卷挂会覆盖整个目录,而 `.conf` 和 `lua/` 同在 `conf.d/`。 - 所以把 `session_route*.conf` 移到子目录 **`conf.d/routes/`**,ConfigMap(`openresty-conf`)只挂到那里; - `lua/` + `router_locations.inc` + `nginx.conf` + **8080 dispatch(`session_base.conf`)** 仍烤镜像。 - `nginx.conf` 的 include 从 `conf.d/*.conf` 改成 `conf.d/routes/*.conf`(`lua_package_path` 不变)。 -- **路径路由(单一对外端口)**: - - 镜像 baked 一个 `listen 8080` 的 dispatch server,按请求路径首段 `//` - 运行时派生到 per-model server 的 unix socket(`/sock/.sock`)—— 单一对外端口、零映射表。 - - per-model server 只 `listen unix:.../.sock`(不占 TCP 端口),故 `spec.nginx.route` = 外部路径 key - = socket 名;autoconfig 生成的 `session_route_.conf` 里就是这个 socket listen。 - - dispatch 把打 8080 的真实客户端 IP 经 `X-Real-IP` 透传,per-model `set_real_ip_from unix:` 还原 - `$remote_addr` → `allow 127.0.0.1` 的调参端点仍只对 in-pod 本地开放。 -- **reload sidecar**:`--process "nginx: master"` + **`--sock-dir`**。 +- **ConfigMap delivery (why a subdirectory)**: mounting a ConfigMap as a whole volume replaces the whole + directory, and the `.conf` files and `lua/` both live in `conf.d/`. So the `session_route*.conf` files moved to + the subdirectory **`conf.d/routes/`**, and the ConfigMap (`openresty-conf`) is mounted only there; `lua/` + + `router_locations.inc` + `nginx.conf` + the **8080 dispatch (`session_base.conf`)** stay baked into the image. + The include in `nginx.conf` changes from `conf.d/*.conf` to `conf.d/routes/*.conf` (`lua_package_path` is + unchanged). +- **Path routing (a single external port)**: + - The image bakes in a dispatch server on `listen 8080` that, by the first path segment `//`, forwards + at runtime to the unix socket of the per-model server (`/sock/.sock`) — one external port, no + mapping table. + - Per-model servers only `listen unix:.../.sock` (no TCP port), so `spec.nginx.route` = external path + key = socket name; the `session_route_.conf` autoconfig generates contains exactly this socket listen. + - The dispatch server passes the real client IP of requests to 8080 in `X-Real-IP`, and the per-model + `set_real_ip_from unix:` restores `$remote_addr` → tuning endpoints restricted with `allow 127.0.0.1` stay + reachable only from inside the pod. +- **reload sidecar**: `--process "nginx: master"` + **`--sock-dir`**. ### cart -- **ConfigMap 交付**:整卷挂 `cart-config` → cart 启动 `-c configs/config.yaml`。底稿(`server`/`cache`/`health`) - 由 CART chart 的 `values.baseConfig` 建在此 ConfigMap,autoconfig 只重填 `workers` 段。 -- **reload sidecar**:`--process cache-aware-router`。 +- **ConfigMap delivery**: `cart-config` is mounted as a whole volume → cart starts with `-c configs/config.yaml`. + The base config (`server`/`cache`/`health`) is created in this ConfigMap by the CART chart's + `values.baseConfig`; autoconfig only rewrites the `workers` section. +- **reload sidecar**: `--process cache-aware-router`. -### 监控 +### Monitoring -- **ConfigMap 交付**:整卷挂 `monitor-conf` → `conf.d/.monitor.conf`(多模型共享,每模型一个 key)。 -- **无 reload sidecar**:自身每 60s 热加载,不需要 SIGHUP(区别于 openresty/cart)。 +- **ConfigMap delivery**: `monitor-conf` is mounted as a whole volume → `conf.d/.monitor.conf` (shared by + many models, one key per model). +- **No reload sidecar**: it reloads itself every 60s and needs no SIGHUP (unlike openresty/cart). -### reload sidecar 接入(openresty / cart 通用) +### Adding the reload sidecar (openresty / cart) -机制见上「reload —— 配置热重载」。接入 = 消费方 pod 加一个 `autoconfig-reload` 容器 -(`shareProcessNamespace: true` 才能发 SIGHUP + 输出 ConfigMap **整卷挂**到 `--watch`),args: +See "reload — config hot reload" above for how it works. To add it, give the consumer pod an +`autoconfig-reload` container (`shareProcessNamespace: true` is needed to send SIGHUP, and the output +ConfigMap must be **mounted as a whole volume** at `--watch`), with args: ```yaml # openresty @@ -321,26 +374,28 @@ args: ["--watch","/watch","--process","nginx: master","--sock-dir","/usr/local/o args: ["--watch","/watch","--process","cache-aware-router"] ``` -## 构建(三个镜像) +## Building (three images) -多阶段 build:golang builder 编译 → 产物打进运行期基础镜像。**不 vendor**; -`go.sum` 入库 + `GOSUMDB=off` 保证可复现,`GOTOOLCHAIN=local` 防联网拉工具链。 +Multi-stage builds: a Go builder compiles, and the binary is copied into a runtime base image. **Nothing is +vendored**; the committed `go.sum` keeps builds reproducible with `GOSUMDB=off`, and `GOTOOLCHAIN=local` stops +Go from downloading a toolchain. -基础镜像与 goproxy 都是 `ARG`,**默认走公网**,clone 下来即可构建: +The base images and the Go module proxy are `ARG`s that **default to public sources**, so a fresh clone builds +as is: ```bash -docker build -t autoconfig:dev . # controller(cmd/) +docker build -t autoconfig:dev . # controller (cmd/) docker build -f Dockerfile.reload -t autoconfig-reload:dev . # reload sidecar docker build -f Dockerfile.hagate -t autoconfig-hagate:dev . # hagate sidecar ``` -| ARG | 默认 | 说明 | +| ARG | Default | Description | |---|---|---| -| `GO_BASE` | `golang:1.23.3-alpine3.20` | 编译阶段基础镜像 | -| `RUNTIME_BASE` | `python:3.12-alpine` | 运行期基础镜像 | -| `GOPROXY` | `https://proxy.golang.org,direct` | 依赖代理 | +| `GO_BASE` | `golang:1.23.3-alpine3.20` | Base image of the build stage | +| `RUNTIME_BASE` | `python:3.12-alpine` | Runtime base image | +| `GOPROXY` | `https://proxy.golang.org,direct` | Go module proxy | -在内网 / 受限网络里换成镜像缓存: +On an internal or restricted network, point them at mirrors: ```bash docker build \ @@ -350,65 +405,71 @@ docker build \ -t autoconfig:dev . ``` -`Makefile` 的 `make docker-build` 已带上这组参数(`BUILD_ARGS` 变量可覆盖,走公网时 -`make docker-build BUILD_ARGS=`)。CI 在打 git tag 时自动构建并推送三个镜像, -chart 的 `version` / `appVersion` 同步成该 tag —— `values.yaml` 的 `image.tag` 留空即回落到 -`appVersion`,**不要在 values 里写死版本号**。 +`make docker-build` passes no overrides by default; pass them through `BUILD_ARGS`, for example +`make docker-build BUILD_ARGS="--build-arg GOPROXY=,direct"`. On a git tag, +`.github/workflows/release.yml` builds the three images (linux/amd64 + linux/arm64) and pushes them to Docker +Hub (`4pdosc/`). The chart is published from +[modelsphere/helm-charts](https://github.com/modelsphere/helm-charts); leave `image.tag` in `values.yaml` empty +so it falls back to the chart's `appVersion` — **do not pin a version in values**. -本地快速验证:`go build ./cmd/...`。 +Quick local check: `go build ./cmd/...`. -## 测试 +## Testing ```bash -make test # 代码生成 + fmt + vet + go test ./... +make test # code generation + fmt + vet + go test ./... ``` -`test/e2e/` 下是端到端脚本,需要一个可用的 k8s 集群(`kubectl` + `helm`),覆盖: -CRD controller 的发现分桶 / 渲染内容 / status / 扩缩跟随 / fail-safe / finalizer 清理, -以及真 openresty + 真 CART 经 helm chart 的接入(真 reload 热更、路径路由 unix socket、 -`openresty -t` 校验、master-standby failover)。脚本默认值用环境变量覆盖, -鉴权 key 等敏感项需显式提供(未设置会直接报错退出,不带默认值)。 +`test/e2e/` holds end-to-end scripts that need a working Kubernetes cluster (`kubectl` + `helm`). They cover +the CRD controller's discovery bucketing / rendered content / status / following scale changes / fail-safe / +finalizer cleanup, and real openresty + real CART integrated through Helm charts (real reload, path routing over +unix sockets, `openresty -t` validation, master-standby failover). Defaults can be overridden with environment +variables; secrets such as the auth key must be provided explicitly (the scripts stop with an error if they are +unset — there is no default). -## 开发布局(kubebuilder / operator-sdk v4) +## Development layout (kubebuilder / operator-sdk v4) -标准 operator 布局:`PROJECT` + `Makefile` + `api/v1alpha1`(带 kubebuilder marker 的类型) -+ `internal/controller`(reconciler)+ `internal/{discovery,sink,hagate,reload}` -+ `config/`(kustomize:crd/rbac/manager/default/samples)。 +The standard operator layout: `PROJECT` + `Makefile` + `api/v1alpha1` (types with kubebuilder markers) ++ `internal/controller` (the reconciler) + `internal/{discovery,sink,hagate,reload}` ++ `config/` (kustomize: crd/rbac/manager/default/samples). -**改了 `api/` 类型或 `+kubebuilder:` marker 后**,跑生成、提交生成物(CI 不跑生成,只编译): +**After changing a type in `api/` or a `+kubebuilder:` marker**, run the generators and commit the output (CI +reruns them and fails on any difference): ```bash -make generate manifests # controller-gen 生成 deepcopy + config/crd/bases + config/rbac/role.yaml,并同步 CRD 到 helm/crds -make test # 生成 + fmt + vet + go test +make generate manifests # controller-gen: deepcopy + config/crd/bases + config/rbac/role.yaml, and copies the CRD into the Helm chart +make test # generate + fmt + vet + go test ``` -工具用 `go run ...@version`(见 Makefile),不装二进制、不入库。 +Tools run as `go run ...@version` (see the Makefile); no binaries are installed or committed. -**部署两条路都可**:生产用 **Helm**(`deploy/helm/autoconfig`);kustomize 用 `make deploy` -(`config/default`)。两者的 CRD/RBAC 同源(都来自 `config/` 的生成物)。 +**Two ways to deploy**: **Helm** for production (`deploy/helm/autoconfig`), or kustomize with `make deploy` +(`config/default`). Both use the same CRD/RBAC, generated from `config/`. -## 注意(踩坑) +## Notes and pitfalls -- **ConfigMap 必须整卷挂**(非 subPath)才会随更新自动同步;kubelet 同步有 **~1min 延迟** —— 对 peer 更新可接受 - (health-timer + proxy_next_upstream 兜过渡),对 CART 反而是天然去抖(reload 会重建 radix tree)。 -- **fail-safe**:发现结果为空绝不写空(CART 拒绝空 workers;openresty 会丢全部流量)。 -- **reload 找 pid 只比 argv[0]**(不是整条 cmdline,也不用 comm —— comm 截断 15 字符):否则 sidecar 自己的 - `--process nginx: master` 参数会自匹配。规则:argv[0] 相等 / basename 相等 / 以 match 开头 - (nginx master 的 argv[0] = `nginx: master process ...`)。 -- **env 名别撞 k8s Service 注入**:若有名为 `cart` 的 Service,k8s 会注入 `CART_PORT=tcp://...`; - autoconfig 的 env 前缀统一 `PS_`。 +- **ConfigMaps must be mounted as whole volumes** (not subPath) to follow updates, and the kubelet syncs them + with a **delay of about a minute** — acceptable for peer updates (health timers + `proxy_next_upstream` cover + the transition), and for CART a natural debounce (a reload rebuilds the radix tree). +- **Fail-safe**: an empty discovery result is never written (CART rejects an empty worker list; openresty would + drop all traffic). +- **reload matches the pid on argv[0] only** (not the whole cmdline, and not comm, which is truncated to 15 + characters): otherwise the sidecar's own `--process nginx: master` argument would match itself. Rule: argv[0] + equal / basename equal / starts with the match (the nginx master's argv[0] is `nginx: master process ...`). +- **Keep env names clear of Kubernetes Service injection**: if there is a Service named `cart`, Kubernetes + injects `CART_PORT=tcp://...`; autoconfig's env prefix is always `PS_`. -### 为什么限流的默认 key 不是 `$binary_remote_addr` +### Why the rate-limit key does not default to `$binary_remote_addr` -生产链路是 dispatch(`:8080` TCP)→ `proxy_pass` 到 **unix socket** → 各路由的 server 块。 -在 unix socket 那一跳上没有 IP,实测路由层拿到的是: +The production path is dispatch (`:8080` TCP) → `proxy_pass` to a **unix socket** → each route's server block. +The unix-socket hop has no IP; measured at the route level: ``` -经 dispatch → unix socket: remote_addr=[unix:] xff=[127.0.0.1] xrealip=[127.0.0.1] -客户端带 XFF 时: remote_addr=[unix:] xff=[203.0.113.7, 127.0.0.1] -直连对照(不经 socket): remote_addr=[127.0.0.1] +via dispatch → unix socket: remote_addr=[unix:] xff=[127.0.0.1] xrealip=[127.0.0.1] +client sends XFF: remote_addr=[unix:] xff=[203.0.113.7, 127.0.0.1] +direct, for comparison: remote_addr=[127.0.0.1] ``` -`$remote_addr` 恒等于字符串 `unix:` —— 每个请求算出同一个 key, -`limit_conn 4` 就成了「整个服务同时只许 4 条」,而不是「每 IP 4 条」。 -这与外层是不是网关无关,是 dispatch→路由走 unix socket 这个结构决定的。 +`$remote_addr` is always the string `unix:` — every request computes the same key, so `limit_conn 4` becomes +"4 connections for the whole service" instead of "4 per IP". This has nothing to do with whether there is a +gateway in front; it follows from the dispatch → route hop going through a unix socket. diff --git a/README.zh-CN.md b/README.zh-CN.md new file mode 100644 index 0000000..a9f2d85 --- /dev/null +++ b/README.zh-CN.md @@ -0,0 +1,417 @@ +# autoconfig + +[English](README.md) | 简体中文 + +**让路由层的后端列表跟着 Kubernetes 自动收敛的 operator。** + +在 k8s 上跑推理服务时,后端 pod 的 IP 会随扩缩容、重启、滚动更新不断变化。而前面的路由组件 +—— openresty、cache-aware-router(CART)、监控 —— 各自维护着一份 peer / worker / 采集目标列表。 +人工同步这几份列表既繁琐又容易漏:扩容了没加进去等于白扩,缩容了没摘掉就是持续打死 IP。 + +autoconfig 用一个 `ModelRoute` 自定义资源描述「一个模型的路由长什么样」,然后: + +1. **发现** —— watch 该模型对应 Service 的 EndpointSlice(或 pod label),得到当前就绪的后端端点; +2. **渲染** —— 生成 openresty 的路由配置、CART 的 workers 列表、监控的采集行; +3. **下发** —— 写进各消费方已有的 ConfigMap,由它们各自的 sidecar 热重载生效。 + +```bash +kubectl apply -f config/samples/modelroute-glm.yaml +kubectl get mr -A # NAME BACKENDS CART READY AGE +``` + +后端扩缩容时不需要任何人工操作,`BACKENDS` 列会自己变。 + +## 它不做什么 + +- **不创建 chart、不创建 ConfigMap** —— 只把内容写进消费方**已有**的 ConfigMap。消费方的部署、 + 初始配置、sidecar 挂载由各自的 chart 负责(见「消费方接入」)。 +- **不代理流量** —— 它是控制面,数据面仍是 openresty / CART。 +- **不管非 LLM 之外的健康语义** —— `modelType: video` 走纯反向代理,不套 token 级限流那一套。 + +## 快速上手 + +```bash +# 1. 部署 controller(chart 含 CRD + RBAC + Deployment) +helm upgrade --install autoconfig deploy/helm/autoconfig -n llm-route --create-namespace + +# 2. 声明一条路由 +kubectl apply -f config/samples/modelroute-glm.yaml + +# 3. 看发现结果 +kubectl get mr -A +kubectl describe mr # status 里有 backends / cartPeers / conditions +``` + +不用 helm 时走 kustomize(同源生成物):`make install`(装 CRD)+ `make deploy`(起 controller)。 + +**⚠️ 卸载顺序:先删 ModelRoute,再 `helm uninstall`。** ModelRoute 带 finalizer +(`routing.modelsphere.dev/cleanup`),要 controller 在跑才能摘。若先 uninstall(删了 controller) +再删 ModelRoute / namespace,ModelRoute 会卡住、拖住 namespace 与 CRD 的删除。 +正确顺序:`kubectl delete mr --all -A` → `helm uninstall`。 +(chart 里用 `modelRoutes` 声明的 ModelRoute 由 helm 托管,uninstall 前会随 release 删除、 +controller 还在 → 自动摘 finalizer,无此问题。) +已卡住的补救:`kubectl patch mr -n --type=merge -p '{"metadata":{"finalizers":[]}}'`。 + +## 工作原理 + +![autoconfig 架构:controller 按 Service 发现后端端点 → 写 openresty / CART / monitor 三个 ConfigMap;openresty/CART 里 reload sidecar 收 SIGHUP 热重载,monitor 自身每 60s 热加载](docs/architecture.png) + +三个独立二进制 / 镜像,各司其职: + +| 组件 | 镜像 | 角色 | +|---|---|---| +| **controller** | `autoconfig`(`cmd/`) | 唯一发现逻辑 + RBAC 一处;watch ModelRoute + EndpointSlice + Pod → 发现 → 渲染 → 写 ConfigMap + status。controller 自身 `replicas>1` 时靠 manager 的 leader 选举保证只有一个在干活。 | +| **reload sidecar** | `autoconfig-reload`(`cmd/reload`) | 跑在消费方 pod 里,watch 挂载的 ConfigMap 文件,变化就 `kill -HUP` 主进程(靠 `shareProcessNamespace`)。CART / openresty 收 SIGHUP 优雅重载。 | +| **hagate sidecar** | `autoconfig-hagate`(`cmd/hagate`) | 消费方 **master-standby**:2 副本都保持 Ready,但只有持 Lease 的 leader 给自己 pod 打 `-active=true` 标签;Service selector 带这个标签 → **只有 leader 进 endpoints**。用标签而非 readiness 门控,standby 不会永久 NotReady 卡住滚动。 | + +三个镜像的 tag 与 chart 版本同线(chart 的 `appVersion` = 镜像 tag),见「构建」。 + +## 两个 sidecar 的实现原理 + +消费方 pod(openresty / CART)里除主容器外各挂两个 autoconfig sidecar:**hagate**(主备门控)+ **reload**(配置热重载)。 +两者都靠 `shareProcessNamespace: true` 与主容器同 pod 协作。 + +### hagate —— master-standby 单活门控 + +**为什么要单活**:openresty / CART 是**有状态**路由器(openresty 有 session 亲和 + `active_conns` 并发计数, +CART 有 prefix-cache radix tree)。多副本同时进 Service endpoints = 缓存被打散、并发计数分裂,路由质量下降。 +所以要 **2 副本主备(master-standby)**:都保持运行,但同一时刻只有一个对外收流量。 + +**为什么不用 readinessProbe 门控**:若让 standby 的 readiness 恒 NotReady 来挡流量,Deployment 滚动时 +`maxUnavailable`/`minReady` 会把「永久 NotReady 的 standby」当成不可用 → 滚动卡死。故改用**标签门控**而非 readiness。 + +**机制**:每 pod 一个 hagate sidecar 参与 Lease `-ha` 的 leader 选举。 +- 只有持 Lease 的 leader 给**自己 pod** 打 `-active=true` 标签; +- Service 的 selector 带这个标签 → **只有 leader 的 pod 进 endpoints**,standby 在池外待命; +- 消费方上游(如 openresty 的 cart 层)走 **Service ClusterIP VIP**,VIP 恒指 active leader → failover/rollout 对上游透明。 + +**level-triggered 自愈**:每 2s 读 pod 实际标签,与「**该不该 active**(= 是 leader **且**本地 app 端口可连)」比对, +不符就纠正 —— 标签被外部误删也能自愈;本地 app 端口连不上时即便是 leader 也主动摘标签(避免把流量导向坏 pod)。 + +**failover**:计划内下线(SIGTERM=删 pod/滚动/驱逐)release Lease → standby ~1-2s 接管**新**流量; +本 pod **保留** active 标签作 **terminating endpoint**(deletionTimestamp),靠 CNI 的 graceful-terminating +把新连接导向 standby、老在途连接留在本 pod 排空(配合主容器优雅停 + grace),**不硬摘标签 → 不 reset 在途连接**, +pod 退出即自动出 endpoints。存活丢主(Lease 续约失败但 pod 没死)才摘标签离开 Service(避免 2-active)。 +硬崩则等 Lease TTL 过期后接管。 + +**监控组件不做 HA**:采集/告警是**自主轮询循环**(不接收外部流量),readiness 门控挡不住重复采集, +单活无意义 → 单例(`replicas: 1` + `Recreate`),chart 不带 hagate。 +openresty/cart 默认开(`replicas: 2` + `ha.enabled: true`)。 + +### reload —— 配置热重载 + +**问题**:autoconfig 改写了 ConfigMap,主进程(nginx / CART)要重读配置才生效,但不能重启(会断在途长流式连接)。 + +**机制**:每消费方 pod 一个 reload sidecar: +- 把输出 ConfigMap **整卷挂**(非 subPath —— subPath 不随 ConfigMap 更新同步)到 `--watch` 目录,用 fsnotify 监听; +- 文件变 → 找主进程 pid(读 `/proc/*/cmdline` 匹配 `nginx: master` / `cache-aware-router`; + **用 cmdline 不用 `comm`** —— comm 截断 15 字符、且要避开 nginx worker)→ `kill -HUP`; +- nginx / CART 收 **SIGHUP 都是优雅重载**:坏配置只 log warning + 保留旧配置,绝不中断在途请求。 + +**传播延迟**:kubelet 同步挂载的 ConfigMap 有 ~1min 延迟(AtomicWriter `..data` 原子软链切换 → +reload 看到的永远是完整文件,不会读到半写)。 + +**openresty 侧 `--sock-dir`**:per-model server 监听 unix socket,模型删除后 nginx 不会自动 unlink +残留 `.sock`;reload 前先删掉「已无 conf 引用」的孤儿 socket。 + +## ModelRoute CRD + +`ModelRoute`(`routing.modelsphere.dev/v1alpha1`),一个模型一个对象,`kubectl apply` 当场 CEL 校验、 +`kubectl get mr` 看发现结果。完整样例见 [`config/samples/modelroute-glm.yaml`](config/samples/modelroute-glm.yaml)。 +下表逐字段说明(✅=必填)。 + +**`spec` 顶层** + +| 字段 | 必填 | 含义 | +|---|---|---| +| `modelType` | 可选 | 这条路由服务的是哪类模型,决定渲染方式。默认 `llm`;`video` = 视频生成,见下 | +| `discovery` | ✅ | 本模型的后端桶发现方式(喂 CART workers / nginx backend / monitor services) | +| `cart` | 可选 | 配了 = autoconfig 管这个 CART;省略 = nginx 直连后端(无 CART 层) | +| `nginx` | ✅ | openresty 路由:渲染 peers → `session_route_.conf` | +| `monitor` | 可选 | 把发现的后端/入口/CART 也写进共享的监控配置 | + +### modelType:一条路由服务哪类模型 + +| 取值 | 渲染成什么 | 适用 | +|---|---|---| +| `llm`(默认) | 走 lua 路由引擎:会话亲和、TTFT/TPS 限流、自适应并发、CART 前置 | `/v1/chat/completions` 这类 token 流式接口 | +| `video` | **纯反向代理**:不解析请求体、不限流;放开超时、关响应缓冲、透传 `Range`、补齐 `X-Forwarded-Host/Proto` | 视频生成:异步建任务 + 轮询 + 大文件下载 | + +为什么视频不能套 LLM 那套:请求体可能是 64MB 的 base64 图(引擎要解析 body 取 model)、 +一条片子要 1~3 分钟才出结果(TTFT/TPS 这类 token 级指标无从谈起)、响应是几十 MB 的视频流 +(响应缓冲会把它憋在内存或磁盘上),而下载接口还要支持断点续传(`Range` 必须原样透传)。 + +`video` 的可用调优项如下(其余会被 CEL 拒,避免「配了以为生效」): + +| `nginx.values` 键 | 默认 | 含义 | +|---|---|---| +| `max_body_size` | `64m` | 请求体上限(I2V 允许 base64 传图) | +| `proxy_timeout` | `3600s` | 读/写超时(生成 + 大文件下载) | +| `connect_timeout` | `10s` | 连后端超时 | +| `rate_limit` | **不配 = 不限速** | **单连接**下载限速,配了才渲染 `limit_rate`。接受 `200Mbps`/`1.5Gbps`(比特口径,自动换算成 nginx 要的字节/秒)或 nginx 原生写法(`25m`/`512k`) | +| `rate_limit_after` | `1m`(仅当配了 `rate_limit`) | 前 N 字节全速。建任务/查询/删除都是几百字节的 JSON,不该被下载限速拖慢 | +| `upload_conn_limit` | 不配 = 不限 | 每 IP 同时在传的连接数(`limit_conn`),超出直接 503 | +| `upload_req_limit` | 不配 = 不限 | 每 IP 请求速率(`limit_req`,nginx 原生写法如 `10r/s`) | +| `upload_req_burst` | 不配 = 无突发 | 配合 `upload_req_limit` 的突发额度 | +| `api_keys` | 不配 = **不鉴权** | 逗号分隔的 Bearer token,与 LLM 路由同一套约定;不匹配返回 401 | +| `auth_public_paths` | `~^/v2/video_generation/[^/]+/content$` | 免鉴权的路径(nginx map 左值)。默认放行下载:`content.url` 交给最终用户,浏览器不带 Authorization 头,而任务 id 是 UUID、相当于一次性能力 URL。置空 = 连下载也要 key | +| `upload_limit_key` | `$http_x_real_ip` | 上面两个 zone 按什么分组。**不能用 `$binary_remote_addr`**,原因见文末 | + +**下载限速必须配合开缓冲**:`proxy_buffering off` 时 `limit_rate` 会被 nginx 完全忽略 +(50MB 实测:静态文件 4.99s / 开缓冲 4.61s / 关缓冲 0.089s,`proxy_limit_rate` 同理)。 +所以配了 `rate_limit` 时模板渲染成 `proxy_buffering on` + `proxy_max_temp_file_size 0` +—— 开缓冲但不落临时文件,缓冲区满即对上游反压;不配限速时仍是 `proxy_buffering off` 边收边发。 + +**上传方向没有字节级限速**:`limit_rate`/`proxy_limit_rate` 都只作用于响应,nginx 没有 +限制请求体读取速率的指令(真要做只能在 lua 里自己读 `ngx.req.socket` 加 sleep,会丢掉 +`proxy_request_buffering` 的现成反压)。所以上传靠三道闸:`max_body_size` 卡单条体积、 +`upload_conn_limit` 卡并发、`upload_req_limit` 卡频率 —— 单个来源的入向带宽 ≈ 并发数 × 单条速率。 + +限速只限**速度不限大小** —— `client_max_body_size` 管的是请求体,和响应无关; +`proxy_buffering off` 也让响应不落临时文件,所以下载的视频多大都行(1GB 按 200Mbps 约 40 秒)。 +`proxy_read_timeout` 限的是两次数据之间的间隔,不是总时长。 + +CEL 还会拒掉 `video` + `cart` / `slo` / `monitor`:前两个是 LLM 专用;监控的探活与告警 +按 LLM 端点设计,指向视频服务只会产生假告警(用 Prometheus 抓服务自己的指标)。 + +peers 的优先级在 `video` 下映射成 nginx 的主用/`backup` 两档:优先级最高的一组主用, +更低的(如 `backend-svc` 这种 VIP 静态兜底)标 `backup`,pod-IP 那层全挂了才顶上。 + +**下发通道与 llm 完全一致**:同样写进 `nginx.outputConfigMap` 的 `session_route_.conf` 键, +同样由 reload sidecar 监听挂载目录 → `SIGHUP` 生效,没有第二条通道。 +(sidecar 还靠 conf 里的 `listen unix:.../.sock;` 判断哪些 socket 仍在用, +video 模板保持同样的 listen 行格式,有用例守着。) + +样例见 [`config/samples/modelroute-minimax-h3.yaml`](config/samples/modelroute-minimax-h3.yaml)。 + +**`spec.discovery`** —— 一桶后端怎么发现 + +| 字段 | 类型 | 默认/约束 | 含义与配置 | +|---|---|---|---| +| `service` | string | 与 `selector` **二选一** | EndpointSlice 发现(推荐);支持 `ns/name` 跨 ns(裸名默认同 ModelRoute 的 ns)→ ModelRoute 可放中心 ns | +| `selector` | string | 与 `service` **二选一** | pod label 发现(没建 Service 的单机/单卡兜底) | +| `port` | int | 可选,省略自动推导 | 后端端口;service 路径从 EndpointSlice 取、selector 路径从 containerPort 取(仅单端口可推) | +| `includeNotReady` | bool | `false` | 默认只取 Ready 端点(排空中端点自动排除);`true` = 含未 Ready | + +**`spec.cart`** —— 省略整段 = 无 CART + +| 字段 | 类型 | 默认/约束 | 含义与配置 | +|---|---|---|---| +| `service` / `selector` | string | **二选一** | CART pod 发现(供 openresty 的 cart source);`service` 支持 `ns/name` | +| `port` | int | 省略推导 | CART 端口(EndpointSlice/containerPort 单端口自动推) | +| `outputConfigMap` | string | ✅ | 写 CART `config.yaml` 的目标 `ns/name`;底稿(server/cache/health)由 CART chart 的 `values.baseConfig` 建在此 CM,autoconfig 只重填 `workers` 段 | +| `maxLoad` | int | `20` | 每 worker 的 `max_load` | + +**`spec.nginx`** —— openresty 路由 + +| 字段 | 类型 | 默认/约束 | 含义与配置 | +|---|---|---|---| +| `route` | string | 省略 = `metadata.name` | 路由短名 = conf 文件名 + openresty dict 名 + `.sock` + 外部路径 key `//`。**字符集必须 ⊆ `[a-z0-9._-]`**(做 dispatch 路径捕获正则;含大写会派生不到 socket → 8080 打不通) | +| `peers` | list | ✅(≥1) | 有序 peer 组(见 `peers[]` 表) | +| `outputConfigMap` | string | ✅ | 输出 ConfigMap `ns/name`(多路由共享,每路由一个 key,finalizer 摘各自 key) | +| `values` | map | 可选 | 任意调优项原样渲染进 lua `register_route` 返回表(key=value)→ 加新调优项无需改代码 | +| `service` | string | 可选 | nginx 入口自身的 Service `ns/name` → 供监控的 `nginx:` 行 + 入口 pod 扩缩事件驱动;端口取 Service 的 dispatch 命名端口 8080 | +| `selector` | string | 可选 | nginx 入口 pod label 发现(没建 Service 兜底;与 `service` 二选一,都配则 `service` 优先) | + +**`spec.nginx.values` 示例 —— 开启动态限流(自适应并发 AIMD)** + +配了 `tps_limit_tps` 即对本路由 opt-in;未显式 `adaptive_cc: "false"` 时,`adaptive_cc` 按全局默认自动开。 +删掉 `values` 即退回不限流。 + +```yaml +spec: + nginx: + route: qwen + service: llm-route/openresty + outputConfigMap: llm-route/openresty-conf + values: # 任意 key 原样渲染进 lua register_route opts(值必须字符串) + tps_limit_tps: "30" # 解码速率下限(tok/s):EWMA 低于它→AIMD 缩并发,高于它→涨(= opt-in 闸门) + # adaptive_cc_min: "10" # 可选:并发下限(不配 = 静态 max × 全局 min_frac 派生) + # ttft_limit_ms: "60000" # 可选:TTFT 软控阈值 + # adaptive_cc: "false" # 可选:显式关自适应,走静态硬熔断 + peers: + - { use: cart, priority: 3, maxConcurrencyFromBackend: true } + - { use: backend, priority: 2, maxConcurrency: 100 } + - { use: backend-svc, priority: 1 } +``` + +生效后 openresty 侧 `GET //_tps_status` 应见 `opt_in=true, adaptive_cc_on=true`。 + +**`spec.nginx.peers[]`** —— 有序分层(数字大=优先,高优层全 banned 才级联到低层) + +| 字段 | 类型 | 默认/约束 | 含义与配置 | +|---|---|---|---| +| `use` | enum | ✅ `cart`\|`backend`\|`backend-svc` | 见下「三档 `use`」 | +| `priority` | int | — | openresty peer 优先级(建议 cart=3、backend=2、backend-svc=1) | +| `maxConcurrency` | int | 省略用 `values.default_max` | 该组所有 peer 的并发上限 | +| `maxConcurrencyFromBackend` | bool | 仅 `use:cart` 有意义 | `true` = cart 并发上限动态 = 后端单实例并发 × 后端数(CART 扇出到 N 后端,容量随扩缩自动变);设了则忽略静态 `maxConcurrency`,且**要求 `backend` 组 `maxConcurrency>0`** 作乘数 | +| `probePath` | string | 省略见右 | openresty 健康探测路径覆盖(GET,状态行含 200=健康否则 ban)。`use:cart` 默认 `/health`(CART 的 `/v1/models` 是缓存端点、worker 全挂也返 200,不能当信号);其余层默认空 = 用 route 的 `health_probe_path` | + +三档 `use`: +- **`cart`**(priority 3)—— CART 上游,走 **CART Service 的 ClusterIP(VIP,非 pod IP)**:CART 是 master-standby, + VIP 恒指 active leader → CART failover/rollout 对 openresty 透明,autoconfig 无需重写。 +- **`backend`**(priority 2)—— 后端 **pod IP**(session 亲和 / least_conn / per-peer 健康的主力层)。 +- **`backend-svc`**(priority 1,可选兜底)—— 后端 **Service 的 ClusterIP(VIP)静态兜底**: + **autoconfig 本身宕 + 后端 rollout** 时 pod-IP 层是死 IP 又没人重写 → 若无兜底会全断; + VIP 由 kube-proxy 维护、不依赖 autoconfig 存活,pod-IP 层全 banned 后级联到它 → + **降级(走 kube-proxy、无亲和)但不全断**。需 `discovery.service`(selector 模式无 VIP,自动跳过)。 + +**`spec.monitor`** —— 可选;监控组件自身每 60s 热加载,**无 reload sidecar**(区别于 nginx/CART)。每模型一个 key,三类行: + +| 字段 | 类型 | 默认/约束 | 含义与配置 | +|---|---|---|---| +| `outputConfigMap` | string | ✅ | 写监控配置的 ConfigMap `ns/name`(多模型共享,每模型一个 key) | +| `model` | string | 省略 = `metadata.name` | `service:` 行的 model 字段(served-model-name) | +| `gpuType` | string | 省略自动推导 | `service:` 行的 gpu_type;省略 = 从后端节点 GPU label `nvidia.com/gpu.product`(GFD)推短名,推不出留空 | +| `nginx` | bool | 默认 `true`(当 `spec.nginx` 配了 service/selector) | 复用 nginx 入口发现写 `nginx:` 行;`false` 关 | +| `router` | bool | 默认 `true`(当配了 `spec.cart`) | 复用 `spec.cart` 发现的 CART pod 写 `router:` 表(`.../workers`);`false` 关 | + +三类输出行格式:`service: \| \| \| `(每后端实例一行)、 +`nginx: - \| http://ip:8080/`、`router: -router- \| http://ip:port/workers`。 + +**CEL 校验(apply 时即报错)**:① `discovery`/`cart` 的 `service` 与 `selector` 必须**恰好一个**; +② `nginx.peers` 用了 `cart` 必须配 `spec.cart`;③ `cart` 用 `maxConcurrencyFromBackend` 必须给 +`backend` 组配 `maxConcurrency>0`。 + +## 消费方接入 + +autoconfig 只负责把配置**写进已有的 ConfigMap**(不创建 chart、不创建 ConfigMap);消费方各自 chart 建初始 +ConfigMap + reload/hagate sidecar + Service 门控,并把对应 ConfigMap 挂进自己的 pod。 +`autoconfig-reload` / `autoconfig-hagate` 两个 sidecar 镜像由本仓构建,消费方跨仓引用。三个消费方一览: + +| 消费方 | autoconfig 写的 ConfigMap → 挂载文件 | reload sidecar | +|---|---|---| +| **openresty** | `openresty-conf` → `conf.d/routes/session_route_.conf` | ✅ `--process "nginx: master"` + `--sock-dir` | +| **cart**(cache-aware-router) | `cart-config` → `configs/config.yaml`(只重填 `workers` 段) | ✅ `--process cache-aware-router` | +| **监控** | `monitor-conf` → `conf.d/.monitor.conf` | ❌ 自身每 60s 热加载 | + +### openresty + +- **ConfigMap 交付(为什么挂子目录)**:ConfigMap 整卷挂会覆盖整个目录,而 `.conf` 和 `lua/` 同在 `conf.d/`。 + 所以把 `session_route*.conf` 移到子目录 **`conf.d/routes/`**,ConfigMap(`openresty-conf`)只挂到那里; + `lua/` + `router_locations.inc` + `nginx.conf` + **8080 dispatch(`session_base.conf`)** 仍烤镜像。 + `nginx.conf` 的 include 从 `conf.d/*.conf` 改成 `conf.d/routes/*.conf`(`lua_package_path` 不变)。 +- **路径路由(单一对外端口)**: + - 镜像 baked 一个 `listen 8080` 的 dispatch server,按请求路径首段 `//` + 运行时派生到 per-model server 的 unix socket(`/sock/.sock`)—— 单一对外端口、零映射表。 + - per-model server 只 `listen unix:.../.sock`(不占 TCP 端口),故 `spec.nginx.route` = 外部路径 key + = socket 名;autoconfig 生成的 `session_route_.conf` 里就是这个 socket listen。 + - dispatch 把打 8080 的真实客户端 IP 经 `X-Real-IP` 透传,per-model `set_real_ip_from unix:` 还原 + `$remote_addr` → `allow 127.0.0.1` 的调参端点仍只对 in-pod 本地开放。 +- **reload sidecar**:`--process "nginx: master"` + **`--sock-dir`**。 + +### cart + +- **ConfigMap 交付**:整卷挂 `cart-config` → cart 启动 `-c configs/config.yaml`。底稿(`server`/`cache`/`health`) + 由 CART chart 的 `values.baseConfig` 建在此 ConfigMap,autoconfig 只重填 `workers` 段。 +- **reload sidecar**:`--process cache-aware-router`。 + +### 监控 + +- **ConfigMap 交付**:整卷挂 `monitor-conf` → `conf.d/.monitor.conf`(多模型共享,每模型一个 key)。 +- **无 reload sidecar**:自身每 60s 热加载,不需要 SIGHUP(区别于 openresty/cart)。 + +### reload sidecar 接入(openresty / cart 通用) + +机制见上「reload —— 配置热重载」。接入 = 消费方 pod 加一个 `autoconfig-reload` 容器 +(`shareProcessNamespace: true` 才能发 SIGHUP + 输出 ConfigMap **整卷挂**到 `--watch`),args: + +```yaml +# openresty +args: ["--watch","/watch","--process","nginx: master","--sock-dir","/usr/local/openresty/nginx/sock"] +# cart +args: ["--watch","/watch","--process","cache-aware-router"] +``` + +## 构建(三个镜像) + +多阶段 build:golang builder 编译 → 产物打进运行期基础镜像。**不 vendor**; +`go.sum` 入库 + `GOSUMDB=off` 保证可复现,`GOTOOLCHAIN=local` 防联网拉工具链。 + +基础镜像与 goproxy 都是 `ARG`,**默认走公网**,clone 下来即可构建: + +```bash +docker build -t autoconfig:dev . # controller(cmd/) +docker build -f Dockerfile.reload -t autoconfig-reload:dev . # reload sidecar +docker build -f Dockerfile.hagate -t autoconfig-hagate:dev . # hagate sidecar +``` + +| ARG | 默认 | 说明 | +|---|---|---| +| `GO_BASE` | `golang:1.23.3-alpine3.20` | 编译阶段基础镜像 | +| `RUNTIME_BASE` | `python:3.12-alpine` | 运行期基础镜像 | +| `GOPROXY` | `https://proxy.golang.org,direct` | 依赖代理 | + +在内网 / 受限网络里换成镜像缓存: + +```bash +docker build \ + --build-arg GO_BASE=/library/golang:1.23.3-alpine3.20 \ + --build-arg RUNTIME_BASE=//python:3.12-alpine \ + --build-arg GOPROXY= \ + -t autoconfig:dev . +``` + +`Makefile` 的 `make docker-build` 默认不带覆盖参数(走公网);需要时用 `BUILD_ARGS` 传入, +如 `make docker-build BUILD_ARGS="--build-arg GOPROXY=,direct"`。打 git tag 时 +`.github/workflows/release.yml` 自动构建三个镜像(linux/amd64 + linux/arm64)并推送到 Docker Hub(`4pdosc/`)。 +chart 发布在 [modelsphere/helm-charts](https://github.com/modelsphere/helm-charts);`values.yaml` 的 `image.tag` +留空即回落到 chart 的 `appVersion`,**不要在 values 里写死版本号**。 + +本地快速验证:`go build ./cmd/...`。 + +## 测试 + +```bash +make test # 代码生成 + fmt + vet + go test ./... +``` + +`test/e2e/` 下是端到端脚本,需要一个可用的 k8s 集群(`kubectl` + `helm`),覆盖: +CRD controller 的发现分桶 / 渲染内容 / status / 扩缩跟随 / fail-safe / finalizer 清理, +以及真 openresty + 真 CART 经 helm chart 的接入(真 reload 热更、路径路由 unix socket、 +`openresty -t` 校验、master-standby failover)。脚本默认值用环境变量覆盖, +鉴权 key 等敏感项需显式提供(未设置会直接报错退出,不带默认值)。 + +## 开发布局(kubebuilder / operator-sdk v4) + +标准 operator 布局:`PROJECT` + `Makefile` + `api/v1alpha1`(带 kubebuilder marker 的类型) ++ `internal/controller`(reconciler)+ `internal/{discovery,sink,hagate,reload}` ++ `config/`(kustomize:crd/rbac/manager/default/samples)。 + +**改了 `api/` 类型或 `+kubebuilder:` marker 后**,跑生成、提交生成物(CI 会重跑生成,有差异即失败): + +```bash +make generate manifests # controller-gen 生成 deepcopy + config/crd/bases + config/rbac/role.yaml,并同步 CRD 到 helm/crds +make test # 生成 + fmt + vet + go test +``` + +工具用 `go run ...@version`(见 Makefile),不装二进制、不入库。 + +**部署两条路都可**:生产用 **Helm**(`deploy/helm/autoconfig`);kustomize 用 `make deploy` +(`config/default`)。两者的 CRD/RBAC 同源(都来自 `config/` 的生成物)。 + +## 注意(踩坑) + +- **ConfigMap 必须整卷挂**(非 subPath)才会随更新自动同步;kubelet 同步有 **~1min 延迟** —— 对 peer 更新可接受 + (health-timer + proxy_next_upstream 兜过渡),对 CART 反而是天然去抖(reload 会重建 radix tree)。 +- **fail-safe**:发现结果为空绝不写空(CART 拒绝空 workers;openresty 会丢全部流量)。 +- **reload 找 pid 只比 argv[0]**(不是整条 cmdline,也不用 comm —— comm 截断 15 字符):否则 sidecar 自己的 + `--process nginx: master` 参数会自匹配。规则:argv[0] 相等 / basename 相等 / 以 match 开头 + (nginx master 的 argv[0] = `nginx: master process ...`)。 +- **env 名别撞 k8s Service 注入**:若有名为 `cart` 的 Service,k8s 会注入 `CART_PORT=tcp://...`; + autoconfig 的 env 前缀统一 `PS_`。 + +### 为什么限流的默认 key 不是 `$binary_remote_addr` + +生产链路是 dispatch(`:8080` TCP)→ `proxy_pass` 到 **unix socket** → 各路由的 server 块。 +在 unix socket 那一跳上没有 IP,实测路由层拿到的是: + +``` +经 dispatch → unix socket: remote_addr=[unix:] xff=[127.0.0.1] xrealip=[127.0.0.1] +客户端带 XFF 时: remote_addr=[unix:] xff=[203.0.113.7, 127.0.0.1] +直连对照(不经 socket): remote_addr=[127.0.0.1] +``` + +`$remote_addr` 恒等于字符串 `unix:` —— 每个请求算出同一个 key, +`limit_conn 4` 就成了「整个服务同时只许 4 条」,而不是「每 IP 4 条」。 +这与外层是不是网关无关,是 dispatch→路由走 unix socket 这个结构决定的。 diff --git a/RUN_BOOK.md b/RUN_BOOK.md deleted file mode 100644 index 9b4d549..0000000 --- a/RUN_BOOK.md +++ /dev/null @@ -1,98 +0,0 @@ -# autoconfig RUN BOOK - -autoconfig 运维手册。当前聚焦:**验证后端扩缩容时,autoconfig 是否把 openresty / cart / monitor 的配置自动同步跟随**。 - -## 组件与版本(k8s-cpu-20 / `llm-route`+`monitoring` ns,2026-08-04) - -| 组件 | helm release | 运行镜像 | -|---|---|---| -| autoconfig controller | `autoconfig-0.3.32` | `autoconfig:0.3.32` | -| openresty | `openresty-0.1.1` | `llm-openresty:0.1.1` + sidecar `autoconfig-reload/hagate:0.3.32` | -| cart | `cache_aware_router-0.1.1` (app `v0.6.2-k8s`) | `cache_aware_router:v0.6.2-k8s` + sidecar `:0.3.32` | -| monitor | `monitor-0.1.0` | `llm-monitor:0.1.0`(在 `monitoring` ns) | - -**链路**:后端 pod 变化 → k8s EndpointSlice → **autoconfig watch → 重渲染 → 写 ConfigMap**(秒级)→ 消费方 pod 挂载卷更新(kubelet 传播 **~1min lag**)→ reload sidecar `SIGHUP` → 生效。 - -autoconfig 写的 3 个 ConfigMap: -- openresty:`llm-route/openresty-conf` key `session_route_.conf`(peers 块) -- cart:`llm-route/cart-config` key `config.yaml`(workers) -- monitor:`monitoring/monitor-conf` key `.monitor.conf`(service/nginx/router 行) - ---- - -## 验证:扩缩容 → 配置自动同步 - -以 `opt` 路由(后端 = `opt` ns 的 vLLM `opt-125m`,route 名 `opt`)为例。**全部命令在 k8s-cpu-20 上跑**(kubectl 在那)。建议开 2-3 个终端:一个执行 scale,其余 `watch` 观察。 - -### 0. 辅助变量(每个新终端先跑一次) -```bash -NS=llm-route -AUTH="Authorization: Bearer ${AUTH_KEY:?需设置:openresty 入口鉴权 key(Bearer)}" -# openresty active leader pod(HA hagate 只有它对外服务 + 有 curl,用它做 exec 探测) -orexec(){ kubectl -n $NS exec "$(kubectl -n $NS get pod -l openresty-active=true -o jsonpath='{.items[0].metadata.name}')" -c openresty -- "$@"; } -``` - -### 1. opt 后端 replica 2 → 1(制造变化) -```bash -kubectl -n opt get pod -l app=opt -o wide # 当前 - -kubectl -n opt scale deploy opt --replicas=2 # 扩到 2 → 触发"新增 peer" -kubectl -n opt rollout status deploy opt - -# 观察一会儿(见下面 2-5)后,缩回 1 → 触发"摘除 peer" -kubectl -n opt scale deploy opt --replicas=1 -kubectl -n opt get pod -l app=opt -o wide -``` - -### 2. autoconfig status 变化(controller 感知) -```bash -# ModelRoute status:backends 应 1→2→1,observedGeneration / lastSyncTime 跟着跳 -watch -n1 "kubectl -n $NS get mr opt -o custom-columns='BACKENDS:.status.backends,CARTPEERS:.status.cartPeers,GEN:.status.observedGeneration,READY:.status.ready,SYNC:.status.lastSyncTime'" - -# (可选)controller 日志看 reconcile 触发 -kubectl -n $NS logs -l app.kubernetes.io/name=autoconfig --tail=30 -f -``` - -### 3. openresty 配置自动更新(**关键对比:ConfigMap 快、pod 生效慢 ~1min**) -```bash -# a) autoconfig 写的 ConfigMap —— 秒级跟随(backend peer 行) -watch -n1 "kubectl -n $NS get cm openresty-conf -o jsonpath='{.data.session_route_opt\.conf}' | grep -E '8000, \"backend-'" - -# b) openresty POD 实际路由用的 peer(_health_status)—— 有 ~1min 挂载卷传播 lag -watch -n2 "kubectl -n $NS exec \$(kubectl -n $NS get pod -l openresty-active=true -o jsonpath='{.items[0].metadata.name}') -c openresty -- curl -s http://127.0.0.1:8080/opt/_health_status" -``` -对比 a / b:scale 后 **a 立刻变(0-1s),b 要等约一个 kubelet ConfigMap 同步周期(实测 ~57s)才变**。这段窗口内 openresty 仍按旧 peer 列表路由(对已摘除的死 pod 靠 `proxy_next_upstream` 重试兜)。 - -### 4. cart 配置自动更新(同样 ConfigMap 快、cart 生效有 lag) -```bash -# a) ConfigMap 里的 workers —— 快 -watch -n1 "kubectl -n $NS get cm cart-config -o jsonpath='{.data.config\.yaml}' | grep -E 'url:'" - -# b) cart 实际加载的 worker —— 有 lag -watch -n2 "kubectl -n $NS exec \$(kubectl -n $NS get pod -l openresty-active=true -o jsonpath='{.items[0].metadata.name}') -c openresty -- curl -s http://cart:8071/workers" -``` - -### 5. monitor 配置自动更新 -```bash -# autoconfig 写的 monitor-conf,opt 的 service 行后端 IP 列表随 replica 变 -watch -n1 "kubectl -n monitoring get cm monitor-conf -o jsonpath='{.data.opt\.monitor\.conf}'" -``` -关注 `service: opt-0 | http://:8000 | opt-125m | A100` 行:scale 到 2 出现两条后端(`opt-0`/`opt-1`),缩回 1 变回一条。monitor 进程按自身 reload 周期(~60s)拉取生效。 - -### 预期总结 -| 对象 | 跟随速度 | -|---|---| -| ModelRoute `status.backends` | 秒级 | -| ConfigMap:openresty-conf / cart-config / monitor-conf | 秒级(autoconfig 直接写) | -| openresty pod 实时 peer(`_health_status`) | **滞后 ~1min**(kubelet ConfigMap→挂载卷传播 + reload sidecar SIGHUP) | -| cart pod 实时 workers(`/workers`) | **滞后 ~1min**(同上) | -| monitor 进程 | ConfigMap 传播 + monitor 自身 ~60s reload | - -> **传播 lag 根因**:消费方读的是**挂载的 ConfigMap 卷**,kubelet 同步有 ~1 分钟延迟;reload sidecar 用 inotify 监听挂载文件,只能等文件变才触发,追不上 kubelet。想压到秒级需让 autoconfig **直连信号**(直接 SIGHUP / API 通知)绕开 ConfigMap 卷。参见 `plans/`(TODO)。 - ---- - -## 附:一键脚本 -`tools/k8s-e2e/`(vllm 部署仓)下有自动化脚本: -- `sigterm_peer_removal_latency.sh` —— in-pod 高频轮询,量 delete pod 后 peer 从 openresty 实时状态消失的延迟。 -- `fullstack_cart_openresty_rollout.sh` —— operator 挂 + 单 replica 滚动下,cart(connect_timeout)+ openresty(跨层重试)协同验证。 diff --git a/config/manager/kustomization.yaml b/config/manager/kustomization.yaml index bd94061..75174e5 100644 --- a/config/manager/kustomization.yaml +++ b/config/manager/kustomization.yaml @@ -1,5 +1,5 @@ resources: - manager.yaml images: -- name: harbor.4pd.io/hardcore-tech/autoconfig - newTag: "0.3.41" +- name: 4pdosc/autoconfig + newTag: "0.4.0" diff --git a/config/manager/manager.yaml b/config/manager/manager.yaml index 3be8bc9..3aab5aa 100644 --- a/config/manager/manager.yaml +++ b/config/manager/manager.yaml @@ -26,7 +26,7 @@ spec: terminationGracePeriodSeconds: 10 containers: - name: manager - image: harbor.4pd.io/hardcore-tech/autoconfig:latest + image: 4pdosc/autoconfig:latest imagePullPolicy: Always args: [] # 默认开 leader 选举;--leader-elect=false 可关 env: diff --git a/config/samples/modelroute-minimax-h3.yaml b/config/samples/modelroute-minimax-h3.yaml index 4e1bcd0..8109e9f 100644 --- a/config/samples/modelroute-minimax-h3.yaml +++ b/config/samples/modelroute-minimax-h3.yaml @@ -64,7 +64,7 @@ spec: # = openresty 的直连对端)。**不要用 $binary_remote_addr** —— 路由挂在 unix socket # 上,那一跳没有 IP,路由层的 $remote_addr 恒为 "unix:",所有请求会落进同一个桶, # "每 IP 限 4 条"变成"整个服务限 4 条"。 - # 经 phanrouter 时 X-Real-IP 是 phanrouter 的 IP,仍不是每个真实客户端; + # 经上游代理时 X-Real-IP 是该代理的 IP,仍不是每个真实客户端; # 要精确到客户端得取 X-Forwarded-For 最左一跳: # upload_limit_key: $http_x_forwarded_for # 前提是外层确实透传了这个头 # 注意:video 不配 cart / slo / monitor —— 前两个是 LLM 专用; diff --git a/config/samples/modelroute-qwen.yaml b/config/samples/modelroute-qwen.yaml index f39d2c4..11df5f9 100644 --- a/config/samples/modelroute-qwen.yaml +++ b/config/samples/modelroute-qwen.yaml @@ -3,8 +3,8 @@ # 前置:cart-qwen 需先用 cache_aware_router helm chart 装(一个模型一个 cart): # helm -n llm-route install cart-qwen \ # --set fullnameOverride=cart-qwen \ -# --set reload.image=harbor.4pd.io/hardcore-tech/autoconfig-reload:0.3.33 \ -# --set ha.image=harbor.4pd.io/hardcore-tech/autoconfig-hagate:0.3.33 +# --set reload.image=4pdosc/autoconfig-reload:0.4.0 \ +# --set ha.image=4pdosc/autoconfig-hagate:0.4.0 # (cart 起始 initContainer 等 cart-qwen-config 有 workers → 本 ModelRoute apply 后 autoconfig 写入即就绪) # autoconfig 据此写 openresty-conf(session_route_qwen.conf)+ cart-qwen-config(workers)。 # 所有 service / outputConfigMap 均为 ns/name 全限定 → controller 全集群 watch,本 ModelRoute 放哪个 ns 都行 diff --git a/config/samples/opt-backend.yaml b/config/samples/opt-backend.yaml index 02507aa..f666548 100644 --- a/config/samples/opt-backend.yaml +++ b/config/samples/opt-backend.yaml @@ -19,7 +19,7 @@ spec: template: metadata: { labels: { app: opt } } spec: - nodeName: ucloud-wlcb-gpu-005 + # nodeName: # pin to the node whose hostPath holds the model terminationGracePeriodSeconds: 3600 # 硬切刀:到点才 SIGKILL,覆盖排空 + 传播窗口 containers: - name: vllm diff --git a/config/samples/qwen-backend.yaml b/config/samples/qwen-backend.yaml index 2178182..392f363 100644 --- a/config/samples/qwen-backend.yaml +++ b/config/samples/qwen-backend.yaml @@ -17,12 +17,12 @@ spec: template: metadata: { labels: { app: qwen } } spec: - nodeName: ucloud-wlcb-gpu-005 + # nodeName: # pin to the node whose hostPath holds the model terminationGracePeriodSeconds: 3600 # sglang 默认排空;硬切刀给足(无需 --shutdown-timeout) runtimeClassName: nvidia containers: - name: sglang - image: harbor.4pd.io/hardcore-tech/sglang:v0.5.10.post1-gm-fix-v0.1 + image: lmsysorg/sglang:v0.5.10.post1 command: ["python3","-m","sglang.launch_server"] # --enable-metrics:sglang 默认不暴露 /metrics(404),不加则 monitor 采不到 KV cache/运行中/等待中(全 "-") args: ["--model-path=/models/Qwen3.5-4B","--served-model-name=qwen","--host=0.0.0.0","--port=8000","--mem-fraction-static=0.5","--trust-remote-code","--enable-metrics"] diff --git a/deploy/helm/autoconfig/Chart.yaml b/deploy/helm/autoconfig/Chart.yaml index d99123a..a949806 100644 --- a/deploy/helm/autoconfig/Chart.yaml +++ b/deploy/helm/autoconfig/Chart.yaml @@ -5,4 +5,6 @@ type: application version: 0.3.46 # chart 版本 appVersion: "0.3.46" # 镜像版本(image.tag 缺省用它) keywords: [routing, openresty, cache-aware-router, service-discovery, operator] -home: https://gitlab.4pd.io/inference-production-stack/autoconfig +home: https://github.com/modelsphere/autoconfig +sources: + - https://github.com/modelsphere/autoconfig diff --git a/deploy/helm/autoconfig/values.yaml b/deploy/helm/autoconfig/values.yaml index 9e8dc5c..a17e406 100644 --- a/deploy/helm/autoconfig/values.yaml +++ b/deploy/helm/autoconfig/values.yaml @@ -1,6 +1,6 @@ # autoconfig controller(ModelRoute)部署参数。 image: - repository: harbor.4pd.io/hardcore-tech/autoconfig + repository: 4pdosc/autoconfig tag: "" # 缺省用 Chart.appVersion(= 发版 tag;CI 打 tag 时 sed appVersion=tag) # ⚠️ 别写死版本号:CI 只 sed Chart.yaml 的 version/appVersion、不碰本文件, # 写死会让 chart 升版后 image.tag 仍停在旧值 → 装出来的 controller 用旧镜像。 diff --git a/internal/controller/modelroute_controller.go b/internal/controller/modelroute_controller.go index ebadc36..8d89206 100644 --- a/internal/controller/modelroute_controller.go +++ b/internal/controller/modelroute_controller.go @@ -515,7 +515,7 @@ func (r *ModelRouteReconciler) modelRoutesForSLO(ctx context.Context, obj client // // ⚠️⚠️ **为什么只报不删。** 曾经这里是直接删的,判据是"没有别的 ModelRoute 声明这个 key"。 // 那个判据不成立:openresty 的 route key 是**面向流量**的,它还有没有人用,取决于 -// phanrouter / 调用方还在不在打 /<旧 route 名>,与 k8s 里有没有对象声明它**毫无关系**。 +// 上游代理 / 调用方还在不在打 /<旧 route 名>,与 k8s 里有没有对象声明它**毫无关系**。 // 把"没有对象声明"当成"没人在用",就是在无人看管的情况下删掉一条可能仍在承接流量的路由: // // 泄漏旧 key → 该路由继续服务,但 peers 陈旧(可能 502) diff --git a/internal/sink/openresty.go b/internal/sink/openresty.go index 55b2e31..a92900c 100644 --- a/internal/sink/openresty.go +++ b/internal/sink/openresty.go @@ -63,8 +63,8 @@ const ( // "每 IP 限 4 条"直接塌成"整个服务限 4 条"(与外层是不是网关无关,是结构决定的)。 // // 改用 $http_x_real_ip:dispatch 转发时设了 X-Real-IP = 它自己的 $remote_addr, - // 也就是 openresty 的直连对端 —— 客户端直连时就是客户端本身,经 phanrouter 时是 - // phanrouter 的 IP。要精确到"每真实客户端",配 upload_limit_key 取 XFF 最左一跳 + // 也就是 openresty 的直连对端 —— 客户端直连时就是客户端本身,经上游代理时是 + // 该代理的 IP。要精确到"每真实客户端",配 upload_limit_key 取 XFF 最左一跳 // (前提是外层确实透传 X-Forwarded-For)。 defaultVideoUploadLimitKey = "$http_x_real_ip" // 免鉴权路径:下载。任务 id 是 UUID,等于一次性能力 URL。 diff --git a/internal/sink/openresty_video_test.go b/internal/sink/openresty_video_test.go index 31d76f4..997abb0 100644 --- a/internal/sink/openresty_video_test.go +++ b/internal/sink/openresty_video_test.go @@ -311,7 +311,7 @@ func TestRenderRoute_VideoUploadLimits(t *testing.T) { } // upload_limit_key 决定两个 zone 按什么分组。默认 $binary_remote_addr 是**直连对端**的 IP; -// 经外层网关(如 phanrouter)进来时所有请求同一个 IP,"每 IP 限制"就塌成"全局限制", +// 经上游代理进来时所有请求同一个 IP,"每 IP 限制"就塌成"全局限制", // 那种部署必须能换成 $http_x_forwarded_for —— 这个用例守住这个可配性。 func TestRenderRoute_VideoUploadLimitKey(t *testing.T) { def, _ := RenderRoute(videoData([]config.Peer{{IP: "10.0.0.1", Port: 8080}}, diff --git a/internal/sink/slo_test.go b/internal/sink/slo_test.go index 015ef68..e545828 100644 --- a/internal/sink/slo_test.go +++ b/internal/sink/slo_test.go @@ -149,7 +149,7 @@ func TestWithSLOMetrics_NoMutate(t *testing.T) { // Extra 与 Raw 出现同名键时,**Raw 必须在后** —— lua 表构造式里后写的赢。 // -// 这条不是形式主义:2026-09-01 在 k8s-cpu-16 上实测过,用户在 nginx.values 里手写 +// 这条不是形式主义:2026-09-01 在测试集群上实测过,用户在 nginx.values 里手写 // ttft_metrics 时,conf 里会真的出现两行同名键(Extra 那份被 luaVal 加了引号变成字符串)。 // 顺序对 → CRD 的 table 生效;顺序反了 → 引擎收到字符串 → validate_metrics 整份丢弃 → // **静默降级成静态阈值**,而 /_ttft_status 的 source 看不出任何异常。 diff --git a/test/e2e/crd_e2e.sh b/test/e2e/crd_e2e.sh index 5d5029d..dcfc0f0 100644 --- a/test/e2e/crd_e2e.sh +++ b/test/e2e/crd_e2e.sh @@ -1,17 +1,17 @@ #!/usr/bin/env bash -# autoconfig CRD controller 端到端(在能 kubectl 的机器上跑,如 k8s-cpu-20)。 +# autoconfig CRD controller 端到端(在能 kubectl 的机器上跑)。 # 装 CRD + controller → mock 后端(Service→EndpointSlice)+ mock CART → ModelRoute # → 验 cart-config workers + openresty CART优先/后端兜底 peers + status → scale 跟随 → 删除清理。 # # 依赖:autoconfig helm chart(默认 $HERE/charts/autoconfig,含 CRD+RBAC+controller)。 -# 用法:NS=ac-e2e IMG=harbor.4pd.io/hardcore-tech/autoconfig:0.3.22 bash crd_e2e.sh [--keep] +# 用法:NS=ac-e2e IMG=4pdosc/autoconfig:0.4.0 bash crd_e2e.sh [--keep] set -uo pipefail HERE="$(cd "$(dirname "$0")" && pwd)" NS=${NS:-ac-e2e} CTRL_NS=${CTRL_NS:-autoconfig} AC_CHART=${AC_CHART:-$HERE/charts/autoconfig} -IMG=${IMG:-harbor.4pd.io/hardcore-tech/autoconfig:0.3.22} -MOCK=${MOCK:-harbor.4pd.io/hardcore-tech/python:3.12-alpine} +IMG=${IMG:-4pdosc/autoconfig:0.4.0} +MOCK=${MOCK:-python:3.12-alpine} KEEP=0; [ "${1:-}" = "--keep" ] && KEEP=1 FAIL=0 say(){ echo -e "\n=== $* ==="; } diff --git a/test/e2e/ha_failover_check.sh b/test/e2e/ha_failover_check.sh index 8d57795..e92a1ce 100644 --- a/test/e2e/ha_failover_check.sh +++ b/test/e2e/ha_failover_check.sh @@ -5,9 +5,9 @@ set -uo pipefail HERE="$(cd "$(dirname "$0")" && pwd)" NS=${NS:-kimi} CHARTS=${CHARTS:-$HERE/charts} -OR_IMG=${OR_IMG:-harbor.4pd.io/hardcore-tech/llm-openresty:0.2.0-routes} -CART_IMG=${CART_IMG:-harbor.4pd.io/hardcore-tech/cache_aware_router:v0.6.0} -ACR=${ACR:-harbor.4pd.io/hardcore-tech/autoconfig-reload:0.3.22} +OR_IMG=${OR_IMG:-4pdosc/llm-openresty:0.1.20} +CART_IMG=${CART_IMG:-4pdosc/cache_aware_router:0.6.5} +ACR=${ACR:-4pdosc/autoconfig-reload:0.4.0} AUTH_KEY=${AUTH_KEY:?需设置:openresty 入口鉴权 key(Bearer)} FAIL=0; say(){ echo -e "\n=== $* ==="; }; ok(){ echo " PASS: $*"; }; bad(){ echo " FAIL: $*"; FAIL=1; } diff --git a/test/e2e/helm_e2e.sh b/test/e2e/helm_e2e.sh index 13e8cd8..1407478 100644 --- a/test/e2e/helm_e2e.sh +++ b/test/e2e/helm_e2e.sh @@ -7,8 +7,8 @@ HERE="$(cd "$(dirname "$0")" && pwd)" CHART=${CHART:-$HERE/autoconfig} NS=${NS:-autoconfig} TNS=${TNS:-ac-helm-e2e} -IMG_TAG=${IMG_TAG:-0.3.22} -MOCK=${MOCK:-harbor.4pd.io/hardcore-tech/python:3.12-alpine} +IMG_TAG=${IMG_TAG:-0.4.0} +MOCK=${MOCK:-python:3.12-alpine} KEEP=0; [ "${1:-}" = "--keep" ] && KEEP=1 FAIL=0 say(){ echo -e "\n=== $* ==="; } diff --git a/test/e2e/prod_real_model_e2e.sh b/test/e2e/prod_real_model_e2e.sh index 3f0e37f..4170771 100644 --- a/test/e2e/prod_real_model_e2e.sh +++ b/test/e2e/prod_real_model_e2e.sh @@ -3,9 +3,9 @@ # 后端 = 真 opt-125m vLLM(复用 e2e-test/vllm-mock-vllm-svc,跨 ns discovery)。真 /v1/completions 推理。 set -uo pipefail NS=${NS:-llm-route}; CTRL_NS=autoconfig; CHARTS=~/prod-e2e/charts -H=harbor.4pd.io/hardcore-tech; TAG=clusterip-test +H=${REGISTRY:-4pdosc}; TAG=${IMG_TAG:-0.4.0} AC=$H/autoconfig; ACR=$H/autoconfig-reload:$TAG; AH=$H/autoconfig-hagate:$TAG -OR_IMG=$H/llm-openresty:e2e0; CART_IMG=$H/cache_aware_router:v0.6.0; MON_IMG=$H/llm-monitor:k8s +OR_IMG=${OR_IMG:-$H/llm-openresty:0.1.20}; CART_IMG=${CART_IMG:-$H/cache_aware_router:0.6.5}; MON_IMG=${MON_IMG:?需设置:llm-monitor 镜像(无公开镜像)} BACKEND_SVC=e2e-test/vllm-mock-vllm-svc; BPORT=8000; MODEL=facebook/opt-125m KEEP=${KEEP:-0}; FAIL=0 AUTH_KEY=${AUTH_KEY:?需设置:openresty 入口鉴权 key(Bearer)} diff --git a/test/e2e/real_helm_e2e.sh b/test/e2e/real_helm_e2e.sh index 919fe0a..bf1302f 100644 --- a/test/e2e/real_helm_e2e.sh +++ b/test/e2e/real_helm_e2e.sh @@ -2,21 +2,21 @@ # 真组件端到端(经 Helm chart):autoconfig controller 驱动 **真 openresty + 真 CART + 真 monitor**, # 三个消费方(openresty/cache_aware_router/monitor)各自 chart 部署(位置见下)(chart 建 ConfigMap 初值,autoconfig 更新)。 # 验证:CART 读 workers + /workers 端点、openresty reload 生效 peers、monitor 消费 service+nginx+router 行、scale 跟随。 -# 在能 kubectl+helm 的机器上跑(如 k8s-cpu-20)。依赖同目录:charts/{autoconfig,openresty,cache_aware_router,monitor}(autoconfig chart 含 CRD+RBAC+controller)。 +# 在能 kubectl+helm 的机器上跑。依赖同目录:charts/{autoconfig,openresty,cache_aware_router,monitor}(autoconfig chart 含 CRD+RBAC+controller)。 # openresty / monitor / cache_aware_router chart 各在其仓(llm-openresty / llm-monitor / cache_aware_router)的 k8s/helm/ —— 跑前拷进 charts/{openresty,monitor,cache_aware_router}。 -# IMG_TAG=0.3.22 bash real_helm_e2e.sh [--keep] +# IMG_TAG=0.4.0 bash real_helm_e2e.sh [--keep] set -uo pipefail HERE="$(cd "$(dirname "$0")" && pwd)" NS=${NS:-ac-helm} CTRL_NS=${CTRL_NS:-autoconfig} -TAG=${IMG_TAG:-0.3.22} +TAG=${IMG_TAG:-0.4.0} CHARTS=${CHARTS:-$HERE/charts} -AC=harbor.4pd.io/hardcore-tech/autoconfig -ACR=harbor.4pd.io/hardcore-tech/autoconfig-reload:$TAG -OR_IMG=${OR_IMG:-harbor.4pd.io/hardcore-tech/llm-openresty:0.2.0-routes} -CART_IMG=${CART_IMG:-harbor.4pd.io/hardcore-tech/cache_aware_router:v0.6.0} -MON_IMG=${MON_IMG:-harbor.4pd.io/hardcore-tech/llm-monitor:0.1.0} -MOCK=${MOCK:-harbor.4pd.io/hardcore-tech/python:3.12-alpine} +AC=${AC:-4pdosc/autoconfig} +ACR=${ACR:-4pdosc/autoconfig-reload:$TAG} +OR_IMG=${OR_IMG:-4pdosc/llm-openresty:0.1.20} +CART_IMG=${CART_IMG:-4pdosc/cache_aware_router:0.6.5} +MON_IMG=${MON_IMG:?需设置:llm-monitor 镜像(无公开镜像)} +MOCK=${MOCK:-python:3.12-alpine} KEEP=0; [ "${1:-}" = "--keep" ] && KEEP=1 FAIL=0 say(){ echo -e "\n=== $* ==="; } diff --git a/test/e2e/real_kimi_e2e.sh b/test/e2e/real_kimi_e2e.sh index 968b67d..fa22280 100644 --- a/test/e2e/real_kimi_e2e.sh +++ b/test/e2e/real_kimi_e2e.sh @@ -5,23 +5,23 @@ # 真发一条 /v1/chat/completions 经 openresty→CART→kimi 拿真实回答。 # 依赖同目录:charts/{autoconfig,openresty,cache_aware_router,monitor}(autoconfig chart 含 CRD+RBAC+controller)。 # openresty / monitor / cache_aware_router chart 各在其仓(llm-openresty / llm-monitor / cache_aware_router)的 k8s/helm/ —— 跑前拷进 charts/{openresty,monitor,cache_aware_router}。 -# IMG_TAG=0.3.22 bash real_kimi_e2e.sh [--keep] +# IMG_TAG=0.4.0 bash real_kimi_e2e.sh [--keep] set -uo pipefail HERE="$(cd "$(dirname "$0")" && pwd)" NS=${NS:-kimi} # 与 kimi LWS 同 ns(discovery 同 ns) CTRL_NS=${CTRL_NS:-autoconfig} -TAG=${IMG_TAG:-0.3.22} +TAG=${IMG_TAG:-0.4.0} CHARTS=${CHARTS:-$HERE/charts} KIMI_SVC=${KIMI_SVC:-kimi-k26-leader} MODEL=${MODEL:-kimi-k2.6} LISTEN=${LISTEN:-18080} AUTH_KEY=${AUTH_KEY:?需设置:openresty 入口鉴权 key(Bearer)} # kimi 后端无鉴权 -AC=harbor.4pd.io/hardcore-tech/autoconfig -ACR=harbor.4pd.io/hardcore-tech/autoconfig-reload:$TAG -OR_IMG=${OR_IMG:-harbor.4pd.io/hardcore-tech/llm-openresty:0.2.0-routes} -CART_IMG=${CART_IMG:-harbor.4pd.io/hardcore-tech/cache_aware_router:v0.6.0} -MON_IMG=${MON_IMG:-harbor.4pd.io/hardcore-tech/llm-monitor:0.1.0} -ALP=${ALP:-harbor.4pd.io/hardcore-tech/python:3.12-alpine} +AC=${AC:-4pdosc/autoconfig} +ACR=${ACR:-4pdosc/autoconfig-reload:$TAG} +OR_IMG=${OR_IMG:-4pdosc/llm-openresty:0.1.20} +CART_IMG=${CART_IMG:-4pdosc/cache_aware_router:0.6.5} +MON_IMG=${MON_IMG:?需设置:llm-monitor 镜像(无公开镜像)} +ALP=${ALP:-python:3.12-alpine} KEEP=0; [ "${1:-}" = "--keep" ] && KEEP=1 FAIL=0 say(){ echo -e "\n=== $* ==="; } diff --git a/test/e2e/single_replica_recovery.sh b/test/e2e/single_replica_recovery.sh index 9bbc238..1c9a528 100644 --- a/test/e2e/single_replica_recovery.sh +++ b/test/e2e/single_replica_recovery.sh @@ -5,7 +5,7 @@ # 前提:kimi ns 有 openresty-conf ConfigMap(autoconfig 已写)。用法:bash single_replica_recovery.sh set -uo pipefail NS=${NS:-kimi} -OR_IMG=${OR_IMG:-harbor.4pd.io/hardcore-tech/llm-openresty:0.2.0-routes} +OR_IMG=${OR_IMG:-4pdosc/llm-openresty:0.1.20} avail(){ local a=$(kubectl -n "$NS" get deploy ortest -o jsonpath='{.status.availableReplicas}' 2>/dev/null); echo "${a:-0}"; } echo "=== 起隔离单副本 ortest(openresty 镜像 + 同款 readinessProbe,挂共享路由 conf)===" diff --git a/test/e2e/test_dynamic.sh b/test/e2e/test_dynamic.sh index 2588ea8..c03ca35 100644 --- a/test/e2e/test_dynamic.sh +++ b/test/e2e/test_dynamic.sh @@ -3,9 +3,9 @@ # 后端用可扩缩 mock(autoconfig 发现行为不需真推理);真推理/GPU 数据在 resilience/opt 用真 opt-125m。 set -uo pipefail NS=${NS:-llm-route}; CTRL_NS=autoconfig; CHARTS=~/prod-e2e/charts -H=harbor.4pd.io/hardcore-tech; TAG=clusterip-test +H=${REGISTRY:-4pdosc}; TAG=${IMG_TAG:-0.4.0} AC=$H/autoconfig; ACR=$H/autoconfig-reload:$TAG; AH=$H/autoconfig-hagate:$TAG -OR_IMG=$H/llm-openresty:e2e0; CART_IMG=$H/cache_aware_router:v0.6.0; MON_IMG=$H/llm-monitor:k8s; MOCK=$H/python:3.12-alpine +OR_IMG=${OR_IMG:-$H/llm-openresty:0.1.20}; CART_IMG=${CART_IMG:-$H/cache_aware_router:0.6.5}; MON_IMG=${MON_IMG:?需设置:llm-monitor 镜像(无公开镜像)}; MOCK=${MOCK:-python:3.12-alpine} KEEP=${KEEP:-0}; FAIL=0 say(){ echo -e "\n=== $* ==="; }; ok(){ echo " PASS: $*"; }; bad(){ echo " FAIL: $*"; FAIL=1; } waiteq(){ local w="$1" d="$2"; shift 2; local i g; for i in $(seq 1 60); do g="$("$@" 2>/dev/null)"; [ "$g" = "$w" ] && { ok "$d = $w"; return 0; }; sleep 3; done; bad "$d: got '$g' want '$w'"; return 1; } diff --git a/test/e2e/verify_crossns_port.sh b/test/e2e/verify_crossns_port.sh index 56ae7f1..786f035 100644 --- a/test/e2e/verify_crossns_port.sh +++ b/test/e2e/verify_crossns_port.sh @@ -2,12 +2,12 @@ # 验证两个新特性(0.3.8 端口自动推导 + 0.3.22 service ns/name 跨 ns 发现): # 在【别的 ns(xns)】建 ModelRoute,discovery.service=kimi/kimi-k26-leader(跨 ns)、【不写 port】, # 验证仍发现 kimi leader、端口自动 = 8050。前提:kimi LWS 在跑、autoconfig controller 已升到目标 tag。 -# 依赖:autoconfig helm chart($HERE/charts/autoconfig)。用法:IMG_TAG=0.3.22 bash verify_crossns_port.sh +# 依赖:autoconfig helm chart($HERE/charts/autoconfig)。用法:IMG_TAG=0.4.0 bash verify_crossns_port.sh set -uo pipefail HERE="$(cd "$(dirname "$0")" && pwd)" -TAG=${IMG_TAG:-0.3.22} +TAG=${IMG_TAG:-0.4.0} CTRL_NS=${CTRL_NS:-autoconfig} -AC=harbor.4pd.io/hardcore-tech/autoconfig +AC=${AC:-4pdosc/autoconfig} AC_CHART=${AC_CHART:-$HERE/charts/autoconfig} NS=xns FAIL=0; ok(){ echo " PASS: $*"; }; bad(){ echo " FAIL: $*"; FAIL=1; } diff --git a/test/e2e/verify_status_error.sh b/test/e2e/verify_status_error.sh index 77869df..1f42780 100644 --- a/test/e2e/verify_status_error.sh +++ b/test/e2e/verify_status_error.sh @@ -1,15 +1,15 @@ #!/usr/bin/env bash # 验证:发现失败(多端口 Service 未显式配 port)时,原因写进 ModelRoute status(DiscoverError), # kubectl describe/get 看得到——而非只进 controller 日志。前提:controller 已升到目标 tag。 -# 依赖:autoconfig helm chart($HERE/charts/autoconfig)。用法:IMG_TAG=0.3.22 bash verify_status_error.sh +# 依赖:autoconfig helm chart($HERE/charts/autoconfig)。用法:IMG_TAG=0.4.0 bash verify_status_error.sh set -uo pipefail HERE="$(cd "$(dirname "$0")" && pwd)" -TAG=${IMG_TAG:-0.3.22} +TAG=${IMG_TAG:-0.4.0} CTRL_NS=${CTRL_NS:-autoconfig} -AC=harbor.4pd.io/hardcore-tech/autoconfig +AC=${AC:-4pdosc/autoconfig} AC_CHART=${AC_CHART:-$HERE/charts/autoconfig} NS=sterr -MOCK=${MOCK:-harbor.4pd.io/hardcore-tech/python:3.12-alpine} +MOCK=${MOCK:-python:3.12-alpine} FAIL=0; ok(){ echo " PASS: $*"; }; bad(){ echo " FAIL: $*"; FAIL=1; } cleanup(){ kubectl delete ns "$NS" --wait=false 2>/dev/null; } trap cleanup EXIT