diff --git a/internal/server/models.go b/internal/server/models.go index 7e8dfc6..701650a 100644 --- a/internal/server/models.go +++ b/internal/server/models.go @@ -32,10 +32,13 @@ type modelIndex struct { } type catalog struct { - sig string - at time.Time - models []string - route map[string]string + sig string + at time.Time + models []string + route map[string]string + // routes is every route in the order asked, with what it answered: the + // Playground lists deployments by route, and two routes may serve one name. + routes []routeModels refreshing bool // partial: a route did not answer. openresty loads a new route up to a // minute after autoconfig writes it, so such a list is re-asked sooner. @@ -132,6 +135,11 @@ func (ix *modelIndex) probe(ctx context.Context, name string, t *backendTarget) c := &catalog{route: map[string]string{}} answered := 0 for i, route := range routes { + rm := routeModels{Route: route, Models: lists[i]} + if errs[i] != nil { + rm.Error = errs[i].Error() + } + c.routes = append(c.routes, rm) if errs[i] != nil { ix.log.Warn("route did not list its models", "backend", name, "route", route, "err", errs[i]) continue @@ -228,6 +236,19 @@ func first(s []string) string { return s[0] } +// routeModels is one route as the Playground sees it: the models it lists, or +// why it listed none -- a route whose engines are not ready yet does not answer. +type routeModels struct { + Route string `json:"route"` + Models []string `json:"models"` + Error string `json:"error,omitempty"` +} + +// routeHeader names the route a request is for. The Playground lists deployments, +// and two of them may serve the same model name; the name alone would always pick +// the first. +const routeHeader = "X-ModelSphere-Route" + // modelList is the OpenAI shape of a model list. func modelList(ids []string) map[string]any { data := make([]map[string]string, 0, len(ids)) @@ -244,6 +265,15 @@ func modelList(ids []string) map[string]any { func (s *Server) routeByModel(w http.ResponseWriter, r *http.Request, b config.Backend, t *backendTarget) *backendTarget { rest := strings.TrimPrefix(r.URL.Path, strings.TrimSuffix(b.Prefix, "/")) if r.Method == http.MethodGet || r.Method == http.MethodHead { + if strings.TrimSuffix(rest, "/") == "/routes" { + c, err := s.models.get(r.Context(), b.Name, t, "") + if err != nil { + writeError(w, http.StatusBadGateway, fmt.Sprintf("backend %s: %v", b.Name, err)) + return nil + } + writeJSON(w, http.StatusOK, map[string]any{"routes": c.routes}) + return nil + } if strings.TrimSuffix(rest, "/") == "/v1/models" { c, err := s.models.get(r.Context(), b.Name, t, "") if err != nil { @@ -255,6 +285,17 @@ func (s *Server) routeByModel(w http.ResponseWriter, r *http.Request, b config.B } } + if want := r.Header.Get(routeHeader); want != "" { + r.Header.Del(routeHeader) + if !slices.Contains(t.routes, want) { + writeError(w, http.StatusBadRequest, fmt.Sprintf("backend %s has no route %q", b.Name, want)) + return nil + } + routed := *t + routed.url.Path = strings.TrimSuffix(t.url.Path, "/") + "/" + want + return &routed + } + body, err := io.ReadAll(http.MaxBytesReader(w, r.Body, config.DefaultMaxBodyBytes)) if err != nil { writeError(w, http.StatusRequestEntityTooLarge, "request body too large or unreadable") diff --git a/internal/server/multiroute_test.go b/internal/server/multiroute_test.go index 1303035..27792ed 100644 --- a/internal/server/multiroute_test.go +++ b/internal/server/multiroute_test.go @@ -209,3 +209,49 @@ func TestLateRouteJoinsTheListSoon(t *testing.T) { time.Sleep(20 * time.Millisecond) } } + +// The Playground lists deployments by route: it needs each route's own models, +// including a route that is not ready, and to send a turn to the route it picked +// even when another route serves the same model name. +func TestPlaygroundByRoute(t *testing.T) { + _, h, gw := multiRouteServer(t) + gw.serve("qwen-b", "qwen-a") + admin := login(t, h, "admin", "admin-pw") + + var body struct { + Routes []routeModels `json:"routes"` + } + rec := do(h, "GET", "/api/llm/routes", admin, "") + if err := json.Unmarshal(rec.Body.Bytes(), &body); err != nil { + t.Fatalf("%d %s: %v", rec.Code, rec.Body.String(), err) + } + got := map[string]routeModels{} + for _, r := range body.Routes { + got[r.Route] = r + } + if r := got["broken"]; r.Error == "" || len(r.Models) != 0 { + t.Fatalf("broken = %+v, want an error and no models", r) + } + if r := got["qwen-b"]; r.Error != "" || strings.Join(r.Models, ",") != "qwen-a" { + t.Fatalf("qwen-b = %+v", r) + } + + byRoute := func(route string) *httptest.ResponseRecorder { + req := httptest.NewRequest("POST", "/api/llm/v1/chat/completions", strings.NewReader(`{"model":"qwen-a"}`)) + req.Header.Set("Authorization", "Bearer "+admin) + req.Header.Set(routeHeader, route) + rec := httptest.NewRecorder() + h.ServeHTTP(rec, req) + return rec + } + // By name alone qwen-a goes to the first route serving it; the header picks. + if v := decode(t, do(h, "POST", "/api/llm/v1/chat/completions", admin, `{"model":"qwen-a"}`)); v["route"] != "qwen-a" { + t.Fatalf("by name went to %v", v) + } + if v := decode(t, byRoute("qwen-b")); v["route"] != "qwen-b" || v["path"] != "/v1/chat/completions" { + t.Fatalf("by route went to %v", v) + } + if rec := byRoute("elsewhere"); rec.Code != http.StatusBadRequest { + t.Fatalf("unknown route: %d %s", rec.Code, rec.Body.String()) + } +} diff --git a/web/src/modules/playground/Chat.tsx b/web/src/modules/playground/Chat.tsx index 20a0500..e29f31d 100644 --- a/web/src/modules/playground/Chat.tsx +++ b/web/src/modules/playground/Chat.tsx @@ -3,7 +3,7 @@ import { Button, Card, CardContent, CardHeader, CardTitle, Label, PageBanner } f import { Eraser, FlaskConical } from "lucide-react"; import { buildPayload } from "@/modules/playground/api"; import { Composer } from "@/modules/playground/components/Composer"; -import { ModelSelect, NoModels, useModels } from "@/modules/playground/components/ModelSelect"; +import { ModelSelect, NoModels, useModels, useTarget } from "@/modules/playground/components/ModelSelect"; import { ParamsPanel } from "@/modules/playground/components/ParamsPanel"; import { Transcript } from "@/modules/playground/components/Transcript"; import { ViewCode } from "@/modules/playground/components/ViewCode"; @@ -17,11 +17,14 @@ export function Chat() { const models = useModels(); const [model, setModel] = useState(""); const [form, setForm] = useState(DEFAULT_FORM); - const params = useMemo(() => toChatParams(model, form), [model, form]); + // model is the picked deployment's route; target is that deployment while it + // is still listed and ready. + const target = useTarget(model); + const params = useMemo(() => toChatParams(target?.model ?? "", form, target?.id), [target?.model, target?.id, form]); const chat = useChat(params); useEffect(() => { - const first = models.data?.[0]?.id; + const first = models.data?.find((m) => m.ready)?.id; if (!model && first) setModel(first); }, [models.data, model]); @@ -34,7 +37,7 @@ export function Chat() { icon={} actions={
- + @@ -216,7 +218,7 @@ function ComparePanel({ slot, index, form, height, onModel, onRemove, onState, r {slot.model ? t("compare.emptyReady") : t("compare.emptyPick")}

} + empty={

{target ? t("compare.emptyReady") : t("compare.emptyPick")}

} /> ); diff --git a/web/src/modules/playground/api.ts b/web/src/modules/playground/api.ts index 4483d30..f687d5e 100644 --- a/web/src/modules/playground/api.ts +++ b/web/src/modules/playground/api.ts @@ -1,10 +1,13 @@ import { apiFetch, ApiError, getT, request } from "@/shell"; import "@/modules/playground/i18n"; import { SSEParser } from "@/modules/playground/sse"; +import { joinTargets, type RouteModels, type Target } from "@/modules/playground/targets"; +import { api as swiss } from "@swiss/lib/api"; // console proxies this prefix to llm-openresty's route (console.yaml backends: // llm), adding the gateway key it holds. The browser never sees that key. -const BASE = "/api/llm/v1"; +const GATEWAY = "/api/llm"; +const BASE = `${GATEWAY}/v1`; export interface Model { id: string; @@ -20,6 +23,9 @@ export type ReasoningEffort = "" | "none" | "minimal" | "low" | "medium" | "high export interface ChatParams { model: string; + // The gateway route the deployment is published on. Two deployments may serve + // the same model name; the route is what picks one. + route?: string; system: string; temperature: number; topP: number; @@ -72,6 +78,17 @@ export const api = { return body.data ?? []; }, + // targets is the deployments the deployment pages list, each joined with what + // the gateway says about its route. One page of the largest size swissd + // serves: a Playground picker is not the place to page through releases. + targets: async (): Promise => { + const [deployments, routes] = await Promise.all([ + swiss.deployments(1, 100), + request<{ routes?: RouteModels[] }>("GET", `${GATEWAY}/routes`), + ]); + return joinTargets(deployments.deployments ?? [], routes.routes ?? []); + }, + // streamChat posts one turn and reports deltas as they arrive. The // conversation id rides X-Session-Id, which the gateway pins to a backend: // every turn of a conversation hits the same engine, so its prefix cache is @@ -91,7 +108,11 @@ export const api = { try { res = await apiFetch(`${BASE}/chat/completions`, { method: "POST", - headers: { "Content-Type": "application/json", "X-Session-Id": sessionId }, + headers: { + "Content-Type": "application/json", + "X-Session-Id": sessionId, + ...(params.route ? { "X-ModelSphere-Route": params.route } : {}), + }, body: JSON.stringify(payload), signal, }); diff --git a/web/src/modules/playground/components/ModelSelect.tsx b/web/src/modules/playground/components/ModelSelect.tsx index 5e63b63..5b5f867 100644 --- a/web/src/modules/playground/components/ModelSelect.tsx +++ b/web/src/modules/playground/components/ModelSelect.tsx @@ -3,24 +3,54 @@ import { Link } from "react-router"; import { Select, SelectContent, SelectItem, SelectTrigger, SelectValue } from "@modelsphere/ui"; import { api } from "@/modules/playground/api"; import { useT } from "@/modules/playground/i18n"; +import type { NotReady, Target } from "@/modules/playground/targets"; +// useModels lists the deployments, ready ones first; a target's id is its route. export function useModels() { - return useQuery({ queryKey: ["playground", "models"], queryFn: api.models, retry: false }); + return useQuery({ queryKey: ["playground", "targets"], queryFn: api.targets, retry: false }); +} + +// useTarget is the deployment an id names, while it is still listed and ready. +export function useTarget(id: string): Target | undefined { + const models = useModels(); + return models.data?.find((m) => m.id === id && m.ready); +} + +function catalogLine(m: Target): string { + if (!m.catalogModel) return ""; + return [m.catalogModel + (m.version ? ` v${m.version}` : ""), m.variant].filter(Boolean).join(" · "); } export function ModelSelect({ id, value, onChange, className }: { id?: string; value: string; onChange: (model: string) => void; className?: string }) { const t = useT(); const models = useModels(); + const list = models.data ?? []; + const anyReady = list.some((m) => m.ready); + const items = list.map((m) => ({ value: m.id, label: m.ready ? `${m.release} · ${m.model}` : m.release })); + const placeholder = models.isLoading + ? t("common:status.loading") + : !list.length + ? t("modelSelect.none") + : anyReady + ? t("modelSelect.placeholder") + : t("modelSelect.noneReady"); return ( // null, not "", is what makes Base UI show the placeholder. - value={value || null} onValueChange={(v) => v && onChange(v)} disabled={!models.data?.length}> + items={items} value={value || null} onValueChange={(v) => v && onChange(v)} disabled={!list.length}> - + - {(models.data ?? []).map((m) => ( - - {m.id} + {list.map((m) => ( + +
+ + {m.release} + {m.ready ? model={m.model} : null} + + {catalogLine(m)} + {m.ready ? null : {reasonText(t, m.reason)}} +
))}
@@ -28,18 +58,55 @@ export function ModelSelect({ id, value, onChange, className }: { id?: string; v ); } -// NoModels says why the list is empty: the gateway answered, but no route on it -// has a model ready to serve yet. A failed request is modelsHint's to explain. +function reasonText(t: ReturnType, r?: NotReady): string { + switch (r?.kind) { + case "noRoute": + return t("modelSelect.reason.noRoute"); + case "notPublished": + return t("modelSelect.reason.notPublished", { route: r.route }); + case "notServing": + return t("modelSelect.reason.notServing", { route: r.route }); + default: + return ""; + } +} + +// NoModels explains a list with nothing to talk to: no deployment yet, or none +// ready -- each not-ready one named with why, linked to its own page. A failed +// request is modelsHint's to explain. export function NoModels({ className }: { className?: string }) { const t = useT(); const models = useModels(); - if (!models.isSuccess || models.data.length > 0) return null; + if (!models.isSuccess || models.data.some((m) => m.ready)) return null; + if (!models.data.length) { + return ( +

+ {t("modelSelect.noneHint")}{" "} + + {t("modelSelect.deploy")} + +

+ ); + } return ( -

- {t("modelSelect.noneHint")}{" "} - - {t("modelSelect.deploy")} - -

+
+

{t("modelSelect.noneReadyHint")}

+
    + {models.data.map((m) => ( +
  • + + {m.release} + + {": "} + {reasonText(t, m.reason)} +
  • + ))} +
+
); } + +// The deployment's page in Model Serving (modules/inferences). +function detailPath(m: Target): string { + return `/inferences/${encodeURIComponent(m.release)}/details?${new URLSearchParams({ namespace: m.namespace })}`; +} diff --git a/web/src/modules/playground/locales/en-US.json b/web/src/modules/playground/locales/en-US.json index 9e51956..9aede16 100644 --- a/web/src/modules/playground/locales/en-US.json +++ b/web/src/modules/playground/locales/en-US.json @@ -11,7 +11,7 @@ "newChat": "New chat", "params": "Parameters", "model": "Model", - "modelsSource": "The gateway's current model list (GET /v1/models).", + "modelsSource": "The deployments in Model Serving; one not ready yet is greyed out with why.", "sessionHint": "A conversation keeps one X-Session-Id: the gateway pins the whole conversation to one inference instance, so the prefix cache stays warm.", "transcript": "Conversation", "emptyTitle": "Pick a model to start chatting.", @@ -52,10 +52,17 @@ "stopped": "Stopped" }, "modelSelect": { - "placeholder": "Select a model", + "placeholder": "Select a deployment", "none": "No model deployed yet", - "noneHint": "No model on the gateway is ready to call yet. Deploy one under Model Deployment; it shows up here once ready.", - "deploy": "Deploy a model" + "noneReady": "No deployment is ready", + "noneHint": "No model is deployed yet. Deploy one under Model Serving; it shows up here once ready.", + "deploy": "Deploy a model", + "noneReadyHint": "There are deployments, but none can chat yet:", + "reason": { + "noRoute": "no gateway route, cannot be called from here", + "notPublished": "route {route} is not on the gateway yet", + "notServing": "route {route} has no ready instance yet" + } }, "params": { "system": "System prompt", diff --git a/web/src/modules/playground/locales/zh-CN.json b/web/src/modules/playground/locales/zh-CN.json index 05486e0..a312b69 100644 --- a/web/src/modules/playground/locales/zh-CN.json +++ b/web/src/modules/playground/locales/zh-CN.json @@ -4,7 +4,7 @@ "newChat": "新建对话", "params": "参数", "model": "模型", - "modelsSource": "网关当前的模型列表(GET /v1/models)。", + "modelsSource": "模型服务里的部署;未就绪的会置灰并注明原因。", "sessionHint": "同一个对话使用固定的 X-Session-Id:网关把整段对话固定在同一个推理实例上,前缀缓存因此保持命中。", "transcript": "对话", "emptyTitle": "选择一个模型开始对话。", @@ -45,10 +45,17 @@ "stopped": "已停止" }, "modelSelect": { - "placeholder": "选择模型", + "placeholder": "选择部署", "none": "还未部署模型", - "noneHint": "网关上还没有可调用的模型。先到模型部署里部署一个,就绪后会出现在这里。", - "deploy": "去部署" + "noneReady": "部署均未就绪", + "noneHint": "还没有部署任何模型。先到模型服务里部署一个,就绪后会出现在这里。", + "deploy": "去部署", + "noneReadyHint": "已有部署,但都还不能对话:", + "reason": { + "noRoute": "未开启网关路由,无法从这里调用", + "notPublished": "路由 {route} 尚未发布到网关", + "notServing": "路由 {route} 暂无就绪实例" + } }, "params": { "system": "系统提示词", diff --git a/web/src/modules/playground/params.ts b/web/src/modules/playground/params.ts index a96f59a..2599b36 100644 --- a/web/src/modules/playground/params.ts +++ b/web/src/modules/playground/params.ts @@ -28,9 +28,10 @@ export const DEFAULT_FORM: ParamsForm = { export const REASONING_EFFORTS: ReasoningEffort[] = ["none", "minimal", "low", "medium", "high"]; -export function toChatParams(model: string, form: ParamsForm): ChatParams { +export function toChatParams(model: string, form: ParamsForm, route?: string): ChatParams { return { model, + route, system: form.system.trim(), temperature: optional(form.temperature), topP: optional(form.topP), diff --git a/web/src/modules/playground/targets.test.ts b/web/src/modules/playground/targets.test.ts new file mode 100644 index 0000000..476ef50 --- /dev/null +++ b/web/src/modules/playground/targets.test.ts @@ -0,0 +1,33 @@ +import { describe, expect, it } from "vitest"; +import type { Deployment } from "@swiss/lib/api"; +import { joinTargets } from "@/modules/playground/targets"; + +const dep = (release: string, route?: string): Deployment => ({ release, namespace: "ns", revision: 1, model: "qwen3.6-35b-a3b", version: "1.1.0", route }); + +describe("joinTargets", () => { + const routes = [ + { route: "qwen-a", models: ["qwen"] }, + { route: "qwen-b", models: ["qwen"] }, + { route: "cold", models: [], error: "GET /cold/v1/models: 502 Bad Gateway" }, + { route: "someone-elses", models: ["other"] }, + ]; + + it("lists deployments only, ready first, each by its own route", () => { + const got = joinTargets([dep("cold", "cold"), dep("a", "qwen-a"), dep("b", "qwen-b")], routes); + expect(got.map((t) => [t.id, t.model, t.ready])).toEqual([ + ["qwen-a", "qwen", true], + ["qwen-b", "qwen", true], + ["cold", "", false], + ]); + }); + + it("says why a deployment cannot be talked to", () => { + const got = joinTargets([dep("cold", "cold"), dep("new", "not-yet"), dep("off")], routes); + expect(got.map((t) => t.reason?.kind)).toEqual(["notServing", "notPublished", "noRoute"]); + expect(got[2].id).toBe("ns/off"); + }); + + it("is empty with no deployment, whatever the gateway serves", () => { + expect(joinTargets([], routes)).toEqual([]); + }); +}); diff --git a/web/src/modules/playground/targets.ts b/web/src/modules/playground/targets.ts new file mode 100644 index 0000000..8602fae --- /dev/null +++ b/web/src/modules/playground/targets.ts @@ -0,0 +1,57 @@ +import type { Deployment } from "@swiss/lib/api"; + +// RouteModels is one gateway route as console's /api/llm/routes reports it: the +// models it lists, or why it listed none. +export interface RouteModels { + route: string; + models?: string[] | null; + error?: string; +} + +// A Target is one deployment as the Playground offers it. id is the route, which +// is what tells two deployments of the same served name apart; model is the name +// the engine answers to, what goes in the request body. +export interface Target { + id: string; + model: string; + release: string; + namespace: string; + // Catalog model, version and variant, as the deployment pages show them. + catalogModel?: string; + version?: string; + variant?: string; + ready: boolean; + reason?: NotReady; +} + +// Why a deployment cannot be talked to yet. +export type NotReady = + // Deployed without a gateway route (modelRoute off): nothing to send a turn to. + | { kind: "noRoute" } + // The route is not in the gateway's route ConfigMap yet. + | { kind: "notPublished"; route: string } + // The route is published but its engines did not list a model: not ready. + | { kind: "notServing"; route: string; error?: string }; + +// joinTargets lists the deployments the deployment pages list, each with what +// the gateway says about its route. Routes no deployment owns are left out: the +// Playground tries what was deployed here, nothing else. Ready ones come first. +export function joinTargets(deployments: Deployment[], routes: RouteModels[]): Target[] { + const byRoute = new Map(routes.map((r) => [r.route, r])); + const out = deployments.map((d): Target => { + const base = { + release: d.release, + namespace: d.namespace, + catalogModel: d.model, + version: d.version, + variant: d.variant, + }; + if (!d.route) return { ...base, id: `${d.namespace}/${d.release}`, model: "", ready: false, reason: { kind: "noRoute" } }; + const r = byRoute.get(d.route); + if (!r) return { ...base, id: d.route, model: "", ready: false, reason: { kind: "notPublished", route: d.route } }; + const served = r.models?.[0]; + if (r.error || !served) return { ...base, id: d.route, model: "", ready: false, reason: { kind: "notServing", route: d.route, error: r.error } }; + return { ...base, id: d.route, model: served, ready: true }; + }); + return out.sort((a, b) => Number(b.ready) - Number(a.ready)); +}