From 1c52d6bc1256e6310f0d293df3d10aca34b4f93e Mon Sep 17 00:00:00 2001 From: yujiezhang-ops Date: Sat, 10 Oct 2026 14:08:36 +0800 Subject: [PATCH] feat: organize multimodal capabilities by capability tabs The Multimodal page listed one block per Provider, so finding who can generate an image meant reading every block. It now has one tab per capability, image and video generation first, and opens the first one an account can provide. Each tab lists the keyed Providers (accounts) that can provide it, the ones the user added first, newest first, then built-ins in catalog order, as on the Providers page. Within a tab: the installed Skill, one row per account, which accounts have no model for it, and a manual model ID entry that can target any account. A failed model listing is reported once above the tabs, and the ffmpeg hint once per tab. BootAgent's protocol converters are left out, as on the Providers page. Co-Authored-By: Claude Opus 5 (1M context) --- .../src/components/MultimodalSection.test.tsx | 131 +++++- frontend/src/components/MultimodalSection.tsx | 394 +++++++++++------- frontend/src/i18n.tsx | 18 +- frontend/src/pages/MultimodalPage.tsx | 5 +- frontend/src/styles/app.css | 25 +- 5 files changed, 388 insertions(+), 185 deletions(-) diff --git a/frontend/src/components/MultimodalSection.test.tsx b/frontend/src/components/MultimodalSection.test.tsx index 0be2e3c5..4d325043 100644 --- a/frontend/src/components/MultimodalSection.test.tsx +++ b/frontend/src/components/MultimodalSection.test.tsx @@ -2,7 +2,7 @@ import { act, fireEvent, render, screen, waitFor, within } from "@testing-librar import { MemoryRouter } from "react-router-dom"; import { beforeEach, describe, expect, it, vi } from "vitest"; -import type { MultimodalDetection, MultimodalProvider } from "../types/api"; +import type { MultimodalDetection, MultimodalProvider, StatusResponse } from "../types/api"; const detectMultimodal = vi.fn<() => Promise>(); const probeMultimodal = vi.fn(); @@ -23,7 +23,7 @@ vi.mock("../backend/api", () => ({ vi.mock("../utils/clipboard", () => ({ copyToClipboard: (text: string) => copyToClipboard(text) })); import { I18nProvider, sourceTranslate } from "../i18n"; -import { describeProbe, groupByCapability, groupInstalled, installLabel, MultimodalSection } from "./MultimodalSection"; +import { accountsFor, describeProbe, groupByCapability, groupInstalled, installLabel, MultimodalSection, orderAccounts } from "./MultimodalSection"; import { EXAMPLE_PROMPTS } from "./multimodalPrompts"; const GENERATABLE = ["openai-chat-vision", "openai-images", "openai-audio-transcriptions", "openai-audio-speech", "frames-chat-vision"]; @@ -68,11 +68,15 @@ function detection(partial: Partial = {}): MultimodalDetect const labels = { "claude-code": "Claude Code", codex: "Codex" }; -function renderSection(onSkillsChanged = vi.fn()) { - render(); +function renderSection(onSkillsChanged = vi.fn(), providers?: StatusResponse["providers"]) { + render(); return onSkillsChanged; } +const builtIn = (name: string, order: number) => ({ name, home: "https://example.com", base_url: "https://example.com/v1", order, has_key: true }); +const custom = (name: string, created: string) => ({ name, home: "", base_url: "https://gateway.example/v1", custom: true, has_key: true, created_at: created }); +const visionOnly = (id: string, name: string): MultimodalProvider => ({ ...gateway, id, name, models: [gateway.models[1]] }); + beforeEach(() => { for (const mock of [detectMultimodal, probeMultimodal, installMultimodal, uninstallSkill, copyToClipboard]) mock.mockClear(); detectMultimodal.mockReset(); @@ -87,6 +91,33 @@ describe("groupByCapability", () => { }); }); +describe("orderAccounts", () => { + it("lists the accounts the user added first, newest first, then built-ins in catalog order", () => { + const ordered = orderAccounts( + [visionOnly("ppio", "PPIO"), visionOnly("novita", "Novita"), visionOnly("old", "Old gateway"), visionOnly("new", "New gateway"), visionOnly("unknown", "Unknown")], + { novita: builtIn("Novita", 3), ppio: builtIn("PPIO", 2), old: custom("Old gateway", "2026-01-01T00:00:00Z"), new: custom("New gateway", "2026-09-01T00:00:00Z") }, + ); + expect(ordered.map((entry) => entry.id)).toEqual(["new", "old", "ppio", "novita", "unknown"]); + }); + + it("leaves out BootAgent's protocol converters", () => { + const ordered = orderAccounts( + [visionOnly("bootagent-converter-chat", "BootAgent Converter chat"), visionOnly("ppio", "PPIO")], + { "bootagent-converter-chat": custom("BootAgent Converter chat", "2026-10-01T00:00:00Z"), ppio: builtIn("PPIO", 2) }, + ); + expect(ordered.map((entry) => entry.id)).toEqual(["ppio"]); + }); + + it("keeps the backend order without status records", () => { + expect(orderAccounts([visionOnly("ppio", "PPIO"), visionOnly("mine", "Mine")]).map((entry) => entry.id)).toEqual(["ppio", "mine"]); + }); + + it("only lists accounts with a model for the capability", () => { + expect(accountsFor([gateway, { ...gateway, id: "empty", models: [] }], "vision").map((account) => account.provider.id)).toEqual(["novita"]); + expect(accountsFor([gateway], "image-generation")).toEqual([]); + }); +}); + describe("recommended models", () => { it("preselects the catalog's recommended model among equally usable ones", () => { const speech = (id: string, recommended: boolean) => ({ id, capabilities: [{ id: "text-to-speech" as const, source: "catalog" as const, adapter: "novita-audio", supported: true, installable: true, probe: "minimal" as const, recommended }] }); @@ -164,7 +195,7 @@ describe("MultimodalSection", () => { detectMultimodal.mockResolvedValue(detection()); probeMultimodal.mockResolvedValue({ providerId: "novita", model: "moonshotai/kimi-k3", capability: "vision", adapter: "openai-chat-vision", status: "available" }); renderSection(); - fireEvent.change(await screen.findByLabelText("图片理解"), { target: { value: "moonshotai/kimi-k3" } }); + fireEvent.change(await screen.findByLabelText("Novita 的图片理解模型"), { target: { value: "moonshotai/kimi-k3" } }); expect(screen.getByText("实测通过后可安装")).toBeTruthy(); // No install button while the evidence is only a guess. expect(screen.queryByRole("button", { name: /^安装$/ })).toBeNull(); @@ -178,8 +209,10 @@ describe("MultimodalSection", () => { detectMultimodal.mockResolvedValue(detection()); probeMultimodal.mockResolvedValue({ providerId: "novita", model: "gpt-image-1", capability: "image-generation", adapter: "openai-images", status: "available" }); renderSection(); - await screen.findByText("Novita"); - fireEvent.change(screen.getByLabelText("能力"), { target: { value: "image-generation" } }); + fireEvent.click(await screen.findByRole("tab", { name: /^图片生成/ })); + expect(screen.getByText("还没有已保存 API Key 的模型服务能做图片生成。")).toBeTruthy(); + // One account: nothing to choose between. + expect(screen.queryByLabelText("模型服务")).toBeNull(); fireEvent.change(screen.getByLabelText("模型 ID"), { target: { value: "gpt-image-1" } }); const manualProbe = within(screen.getByText("手动填写模型 ID").closest("details") as HTMLElement).getByRole("button", { name: /^实测$/ }); expect(manualProbe.getAttribute("title")).toContain("发送前会先确认"); @@ -248,12 +281,13 @@ describe("MultimodalSection", () => { await waitFor(() => expect(installMultimodal).toHaveBeenCalledWith({ provider: "novita", model: "qwen/qwen3-vl-8b", capability: "vision", agents: [], guide: true })); }); - it.each(["network", "key", "transient"])("shows a manual model's %s result despite an existing capability row", async (reason) => { + it.each(["network", "key", "transient"])("shows a manual model's %s result beside an account row for the capability", async (reason) => { detectMultimodal.mockResolvedValue(detection({ providers: [{ ...gateway, models: [{ id: "known-whisper", capabilities: [{ id: "speech-to-text", source: "catalog", adapter: "openai-audio-transcriptions", supported: true, installable: true, probe: "minimal" }] }] }] })); probeMultimodal.mockResolvedValue({ providerId: "novita", model: "my-whisper", capability: "speech-to-text", adapter: "openai-audio-transcriptions", status: reason === "key" ? "unavailable" : "unverified", reason }); renderSection(); - await screen.findByText("Novita"); - fireEvent.change(screen.getByLabelText("能力"), { target: { value: "speech-to-text" } }); + // The only capability on offer is the one opened by default. + expect(await screen.findByRole("tab", { name: /^语音识别/, selected: true })).toBeTruthy(); + expect(screen.getByText("known-whisper")).toBeTruthy(); fireEvent.change(screen.getByLabelText("模型 ID"), { target: { value: "my-whisper" } }); const manual = screen.getByText("手动填写模型 ID").closest("details") as HTMLElement; fireEvent.click(within(manual).getByRole("button", { name: /^实测$/ })); @@ -264,8 +298,7 @@ describe("MultimodalSection", () => { detectMultimodal.mockResolvedValue(detection({ providers: [{ ...gateway, models: [] }] })); probeMultimodal.mockRejectedValue(new Error("manual request failed")); renderSection(); - await screen.findByText("Novita"); - fireEvent.change(screen.getByLabelText("能力"), { target: { value: "speech-to-text" } }); + fireEvent.click(await screen.findByRole("tab", { name: /^语音识别/ })); fireEvent.change(screen.getByLabelText("模型 ID"), { target: { value: "my-whisper" } }); const manual = screen.getByText("手动填写模型 ID").closest("details") as HTMLElement; fireEvent.click(within(manual).getByRole("button", { name: /^实测$/ })); @@ -303,7 +336,8 @@ describe("MultimodalSection", () => { it("says which capabilities a Provider documents no API for", async () => { detectMultimodal.mockResolvedValue(detection({ providers: [{ ...gateway, unsupported: ["video-generation"] }] })); renderSection(); - expect(await screen.findByText("视频生成:该模型服务目前没有可核实的公开接口,暂不支持。")).toBeTruthy(); + fireEvent.click(await screen.findByRole("tab", { name: /^视频生成/ })); + expect(screen.getByText("Novita:该模型服务目前没有可核实的公开接口,暂不支持视频生成。")).toBeTruthy(); }); it("offers install without a probe for a catalog video model", async () => { @@ -312,9 +346,9 @@ describe("MultimodalSection", () => { providers: [{ ...gateway, models: [{ id: "wan2.7-t2v", capabilities: [{ id: "video-generation", source: "catalog", adapter: "novita-async", supported: true, installable: true, probe: "", recommended: true }] }] }], })); renderSection(); - expect(await screen.findByLabelText("视频生成")).toBeTruthy(); + expect(await screen.findByRole("tab", { name: /^视频生成/, selected: true })).toBeTruthy(); expect(screen.getByText("内置目录")).toBeTruthy(); - const row = screen.getByLabelText("视频生成").closest("li") as HTMLElement; + const row = screen.getByText("wan2.7-t2v").closest("li") as HTMLElement; expect(within(row).queryByRole("button", { name: /^实测$/ })).toBeNull(); expect(within(row).getByRole("button", { name: /^安装$/ })).toBeTruthy(); }); @@ -337,10 +371,11 @@ describe("MultimodalSection", () => { expect(copyToClipboard).toHaveBeenCalledWith(expect.stringContaining("先读 /Users/me/.bootagent/multimodal/vision/SKILL.md")); }); - it("warns when ffmpeg is missing for video by frames", async () => { - detectMultimodal.mockResolvedValue(detection({ ffmpeg: false, providers: [{ ...gateway, models: [{ id: "qwen/qwen3-vl-8b", capabilities: [{ id: "video-understanding", source: "declared", adapter: "frames-chat-vision", supported: true, installable: true, probe: "minimal" }] }] }] })); + it("warns once per tab when ffmpeg is missing for video by frames", async () => { + const frames: MultimodalProvider = { ...gateway, models: [{ id: "qwen/qwen3-vl-8b", capabilities: [{ id: "video-understanding", source: "declared", adapter: "frames-chat-vision", supported: true, installable: true, probe: "minimal" }] }] }; + detectMultimodal.mockResolvedValue(detection({ ffmpeg: false, providers: [frames, { ...frames, id: "ppio", name: "PPIO" }] })); renderSection(); - expect(await screen.findByText(/没有在 PATH 上找到 ffmpeg/)).toBeTruthy(); + expect(await screen.findAllByText(/没有在 PATH 上找到 ffmpeg/)).toHaveLength(1); }); it("points to the Providers page when no key is saved", async () => { @@ -349,9 +384,67 @@ describe("MultimodalSection", () => { expect(await screen.findByText("去添加模型服务")).toBeTruthy(); }); + it("puts generation first and opens the first capability an account can provide", async () => { + detectMultimodal.mockResolvedValue(detection()); + renderSection(); + const tabs = await screen.findAllByRole("tab"); + expect(tabs.map((tab) => tab.textContent)).toEqual(["图片生成", "视频生成", "图片理解1", "视频理解1", "语音合成", "语音识别"]); + expect(screen.getByRole("tab", { name: /^图片理解/ })).toHaveAttribute("aria-selected", "true"); + expect(screen.getByRole("tabpanel")).toHaveAttribute("aria-labelledby", "multimodal-tab-vision"); + }); + + it("opens image generation by default when an account can do it", async () => { + const images = { id: "gpt-image-1", capabilities: [{ id: "image-generation" as const, source: "probe" as const, adapter: "openai-images", supported: true, installable: true, probe: "confirm" as const }] }; + detectMultimodal.mockResolvedValue(detection({ providers: [{ ...gateway, models: [...gateway.models, images] }] })); + renderSection(); + expect(await screen.findByRole("tab", { name: /^图片生成/, selected: true })).toBeTruthy(); + expect(screen.getByText("gpt-image-1")).toBeTruthy(); + expect(screen.queryByText("还没有已保存 API Key 的模型服务能做图片生成。")).toBeNull(); + }); + + it("switches capability by click and arrow keys", async () => { + detectMultimodal.mockResolvedValue(detection()); + renderSection(); + const vision = await screen.findByRole("tab", { name: /^图片理解/ }); + fireEvent.keyDown(vision, { key: "ArrowRight" }); + expect(screen.getByRole("tab", { name: /^视频理解/ })).toHaveAttribute("aria-selected", "true"); + expect(document.activeElement).toBe(screen.getByRole("tab", { name: /^视频理解/ })); + fireEvent.keyDown(document.activeElement as HTMLElement, { key: "Home" }); + expect(screen.getByRole("tab", { name: /^图片生成/ })).toHaveAttribute("aria-selected", "true"); + fireEvent.click(screen.getByRole("tab", { name: /^图片理解/ })); + expect(screen.getByText("qwen/qwen3-vl-8b", { selector: "option" })).toBeTruthy(); + }); + + it("lists the account the user added before built-in ones and marks it", async () => { + detectMultimodal.mockResolvedValue(detection({ providers: [visionOnly("ppio", "PPIO"), visionOnly("mine", "我的网关")] })); + renderSection(vi.fn(), { ppio: builtIn("PPIO", 2), mine: custom("我的网关", "2026-09-01T00:00:00Z") }); + await screen.findByRole("tab", { name: /^图片理解2/ }); + const accounts = screen.getByRole("tabpanel").querySelectorAll(".multimodal-account"); + expect([...accounts].map((item) => item.querySelector("strong")?.textContent)).toEqual(["我的网关", "PPIO"]); + expect(within(accounts[0] as HTMLElement).getByText("用户添加")).toBeTruthy(); + expect(within(accounts[1] as HTMLElement).queryByText("用户添加")).toBeNull(); + }); + + it("names the accounts that have no model for the open capability", async () => { + detectMultimodal.mockResolvedValue(detection({ providers: [gateway, { ...gateway, id: "ppio", name: "PPIO", models: [] }] })); + renderSection(); + expect(await screen.findByText("PPIO 的模型列表里没有能做图片理解的模型。")).toBeTruthy(); + // With two accounts, the manual entry asks which one to probe on. + fireEvent.click(screen.getByRole("tab", { name: /^图片生成/ })); + expect(screen.getByLabelText("模型服务")).toBeTruthy(); + }); + + it("marks the tab of an installed capability", async () => { + detectMultimodal.mockResolvedValue(detection({ installed: [ + { skillId: "bootagent-vision", capability: "vision", providerId: "novita", model: "qwen/qwen3-vl-8b", adapter: "openai-chat-vision", agents: ["codex"] }, + ] })); + renderSection(); + expect(await screen.findByRole("tab", { name: /^图片理解已安装/ })).toBeTruthy(); + }); + it("reports a failed listing without hiding the Provider", async () => { detectMultimodal.mockResolvedValue(detection({ providers: [{ id: "deepseek", name: "DeepSeek", keyEnv: "BOOTAGENT_DEEPSEEK_API_KEY", keyFile: "/Users/me/.bootagent/skill-env/deepseek.env", unsupported: [], listed: false, message: "API key was rejected (401).", errorCode: "API_KEY_REJECTED", models: [] }] })); renderSection(); - expect(await screen.findByText(/无法读取模型列表/)).toBeTruthy(); + expect(await screen.findByText("DeepSeek:无法读取模型列表:API key was rejected (401).")).toBeTruthy(); }); }); diff --git a/frontend/src/components/MultimodalSection.tsx b/frontend/src/components/MultimodalSection.tsx index 6173e4e7..899b5862 100644 --- a/frontend/src/components/MultimodalSection.tsx +++ b/frontend/src/components/MultimodalSection.tsx @@ -1,10 +1,12 @@ import { Activity, Check, Copy, Download, Images, RefreshCw, Trash2, X } from "lucide-react"; -import { useCallback, useEffect, useMemo, useState } from "react"; +import { useCallback, useEffect, useMemo, useState, type KeyboardEvent } from "react"; import { Link } from "react-router-dom"; import { api, describeError } from "../backend/api"; import type { Translate, TranslationKey } from "../i18n"; import { useI18n } from "../i18n"; +import { isConverterID } from "../state/conversion"; +import { byProviderCreatedAt } from "../state/ranking"; import type { InstalledMultimodal, MultimodalCapability, @@ -15,18 +17,21 @@ import type { MultimodalProbeKind, MultimodalProvider, MultimodalSource, + StatusResponse, } from "../types/api"; import { copyToClipboard } from "../utils/clipboard"; import { ModalDialog } from "./ModalDialog"; import { EXAMPLE_PROMPTS } from "./multimodalPrompts"; -const CAPABILITY_ORDER: MultimodalCapabilityId[] = [ - "vision", +// Also the tab order. Generation leads: it is what users come to this page +// for, while understanding is something most chat models already do. +export const CAPABILITY_ORDER: MultimodalCapabilityId[] = [ "image-generation", - "speech-to-text", - "text-to-speech", "video-generation", + "vision", "video-understanding", + "text-to-speech", + "speech-to-text", ]; export const CAPABILITY_LABELS: Record = { @@ -38,6 +43,14 @@ export const CAPABILITY_LABELS: Record = "video-understanding": "视频理解", }; +// Shown in the manual entry hint: model IDs a user is likely to know. +const MANUAL_EXAMPLES: Partial> = { + vision: "gpt-4o-mini", + "image-generation": "gpt-image-1", + "speech-to-text": "whisper-1", + "text-to-speech": "tts-1", +}; + // Strongest evidence first. An ID rule is a guess from the name, so it sorts // last and is never preselected over anything the Provider or a probe said. const SOURCE_RANK: Record = { probe: 0, declared: 1, catalog: 2, "id-rule": 3 }; @@ -80,6 +93,31 @@ export function groupByCapability(entry: MultimodalProvider): CapabilityGroup[] })); } +/** + * Orders the keyed Providers the way the Providers page lists accounts: the + * ones the user added first, newest first, then built-ins in catalog order. + * The backend lists built-ins first; a Provider the status does not know keeps + * its place after the known ones. + */ +export function orderAccounts(providers: MultimodalProvider[], status?: StatusResponse["providers"]): MultimodalProvider[] { + const rank = new Map(byProviderCreatedAt(status ?? {}).map(([id], index) => [id, index])); + return providers + // BootAgent's own protocol converters hold a key but are not accounts; + // the Providers page hides them too. + .filter((entry) => !isConverterID(entry.id)) + .map((entry, index) => ({ entry, index })) + .sort((left, right) => (rank.get(left.entry.id) ?? rank.size + left.index) - (rank.get(right.entry.id) ?? rank.size + right.index)) + .map(({ entry }) => entry); +} + +/** One capability's accounts: every keyed Provider with a model for it, in account order. */ +export function accountsFor(providers: MultimodalProvider[], capability: MultimodalCapabilityId): Array<{ provider: MultimodalProvider; options: CapabilityOption[] }> { + return providers.flatMap((provider) => { + const group = groupByCapability(provider).find((item) => item.id === capability); + return group ? [{ provider, options: group.options }] : []; + }); +} + export function sourceLabel(capability: MultimodalCapability, t: Translate): string { switch (capability.source) { case "probe": @@ -159,36 +197,43 @@ export function groupInstalled(installed: InstalledMultimodal[]): Array<{ skillI .map(([skillId, variants]) => ({ skillId, capability, variants, agents: [...new Set(variants.flatMap((variant) => variant.agents))].sort() }))); } +// `key` is where the result is shown: an account's row, or a tab's manual entry. type ProbeState = { model: string; checking: boolean; result?: MultimodalProbe; error?: string }; -type ProbeTarget = { provider: MultimodalProvider; capability: MultimodalCapabilityId; model: string; adapter: string; kind: MultimodalProbeKind }; +type ProbeTarget = { key: string; provider: MultimodalProvider; capability: MultimodalCapabilityId; model: string; adapter: string; kind: MultimodalProbeKind }; type InstallTarget = { provider: MultimodalProvider; capability: MultimodalCapabilityId; model: string }; const rowKey = (provider: string, capability: string) => `${provider}::${capability}`; +const manualKey = (capability: string) => `manual::${capability}`; interface MultimodalSectionProps { /** Agent ID to display name. */ agentLabels?: Record; + /** The status's Provider records: account order and which ones the user added. */ + providers?: StatusResponse["providers"]; /** Called after an install or uninstall changed the Skill library. */ onSkillsChanged?: () => void; } /** - * The Skills page's Multimodal capabilities section. Opening it lists models, - * which is free; nothing billed is sent until the user presses a check button, - * and a check that produces billed output asks first. + * The Multimodal page. Each Provider with a saved key is an account; one tab + * per capability lists the accounts that can provide it, the ones the user + * added first. Opening it lists models, which is free; nothing billed is sent + * until the user presses a check button, and a check that produces billed + * output asks first. */ -export function MultimodalSection({ agentLabels = {}, onSkillsChanged }: MultimodalSectionProps) { +export function MultimodalSection({ agentLabels = {}, providers: statusProviders, onSkillsChanged }: MultimodalSectionProps) { const { t, locale } = useI18n(); const [detection, setDetection] = useState(null); const [loading, setLoading] = useState(false); const [error, setError] = useState(""); + const [active, setActive] = useState(null); const [selected, setSelected] = useState>({}); const [probes, setProbes] = useState>({}); const [confirmProbe, setConfirmProbe] = useState(null); const [installTarget, setInstallTarget] = useState(null); const [uninstallTarget, setUninstallTarget] = useState<{ skillId: string; capability: MultimodalCapabilityId; agents: string[] } | null>(null); const [uninstalling, setUninstalling] = useState(false); - const [manual, setManual] = useState>({}); + const [manual, setManual] = useState>>({}); const nameOf = (agentId: string) => agentLabels[agentId] ?? agentId; const separator = locale === "en" ? ", " : "、"; @@ -204,18 +249,17 @@ export function MultimodalSection({ agentLabels = {}, onSkillsChanged }: Multimo useEffect(() => { void detect(); }, [detect]); const runProbe = (target: ProbeTarget, confirmed: boolean) => { - const key = rowKey(target.provider.id, target.capability); - setProbes((current) => ({ ...current, [key]: { model: target.model, checking: true } })); + setProbes((current) => ({ ...current, [target.key]: { model: target.model, checking: true } })); void api.probeMultimodal(target.provider.id, target.model, target.capability, target.adapter, confirmed) .then((result) => { - setProbes((current) => ({ ...current, [key]: { model: target.model, checking: false, result } })); - // Show the probed model in its row; a conclusive verdict is cached by - // the backend and outranks every other source once re-read. - setSelected((current) => ({ ...current, [key]: target.model })); + setProbes((current) => ({ ...current, [target.key]: { model: target.model, checking: false, result } })); + // Show the probed model in its account's row; a conclusive verdict is + // cached by the backend and outranks every other source once re-read. + setSelected((current) => ({ ...current, [rowKey(target.provider.id, target.capability)]: target.model })); return detect(); }) .catch((reason: unknown) => { - setProbes((current) => ({ ...current, [key]: { model: target.model, checking: false, error: describeError(reason, t("无法向模型服务发送验证请求。")).message } })); + setProbes((current) => ({ ...current, [target.key]: { model: target.model, checking: false, error: describeError(reason, t("无法向模型服务发送验证请求。")).message } })); }); }; const probe = (target: ProbeTarget) => (target.kind === "confirm" ? setConfirmProbe(target) : runProbe(target, false)); @@ -234,18 +278,166 @@ export function MultimodalSection({ agentLabels = {}, onSkillsChanged }: Multimo .finally(() => setUninstalling(false)); }; - const providers = useMemo(() => detection?.providers ?? [], [detection]); - const groupsByProvider = useMemo(() => new Map(providers.map((entry) => [entry.id, groupByCapability(entry)])), [providers]); - const providerName = (id: string) => providers.find((entry) => entry.id === id)?.name || id; - const generatable = detection?.generatable ?? []; + const accounts = useMemo(() => orderAccounts(detection?.providers ?? [], statusProviders), [detection, statusProviders]); + const byCapability = useMemo(() => new Map(CAPABILITY_ORDER.map((id) => [id, accountsFor(accounts, id)])), [accounts]); const installedCards = useMemo(() => groupInstalled(detection?.installed ?? []), [detection]); + const providerName = (id: string) => accounts.find((entry) => entry.id === id)?.name || statusProviders?.[id]?.name || id; + const generatable = detection?.generatable ?? []; + // Until the user picks a tab: the first capability something can be done with. + const current = active + ?? CAPABILITY_ORDER.find((id) => byCapability.get(id)?.length || installedCards.some((card) => card.capability === id)) + ?? CAPABILITY_ORDER[0]; + + const onTabKey = (event: KeyboardEvent, index: number) => { + if (!["ArrowLeft", "ArrowRight", "Home", "End"].includes(event.key)) return; + event.preventDefault(); + const last = CAPABILITY_ORDER.length - 1; + const next = CAPABILITY_ORDER[event.key === "Home" ? 0 : event.key === "End" ? last : (index + (event.key === "ArrowRight" ? 1 : -1) + CAPABILITY_ORDER.length) % CAPABILITY_ORDER.length]; + setActive(next); + document.getElementById(`multimodal-tab-${next}`)?.focus(); + }; + const renderPanel = (capability: MultimodalCapabilityId) => { + const capabilityName = t(CAPABILITY_LABELS[capability]); + const offering = byCapability.get(capability) ?? []; + const card = installedCards.find((item) => item.capability === capability); + const unsupported = accounts.filter((entry) => entry.unsupported.includes(capability)); + const without = accounts.filter((entry) => entry.listed && !entry.unsupported.includes(capability) && !offering.some((account) => account.provider.id === entry.id)); + const manualRoute = (detection?.manual ?? []).find((route) => route.capability === capability); + const manualState = manual[capability]; + const manualProvider = accounts.find((entry) => entry.id === manualState?.provider) ?? accounts[0]; + const manualProbe = probes[manualKey(capability)]; + const example = MANUAL_EXAMPLES[capability]; + const chosenFor = (entry: MultimodalProvider, options: CapabilityOption[]) => options.find((option) => option.model === selected[rowKey(entry.id, capability)]) ?? options[0]; + // Said once for the tab, not on every account that samples frames. + const needsFFmpeg = Boolean(detection && !detection.ffmpeg) && offering.some(({ provider: entry, options }) => chosenFor(entry, options).capability.adapter === "frames-chat-vision"); + return ( + <> + {card ? ( +
+
+ {t("已安装")} + {card.skillId} + +
+ {card.variants.map((variant) => ( +

+ {variant.agents.length + ? t("{model}({provider})已安装到 {agents}", { model: variant.model, provider: providerName(variant.providerId), agents: variant.agents.map(nameOf).join(separator) }) + : t("{model}({provider})", { model: variant.model, provider: providerName(variant.providerId) })} +

+ ))} + {card.variants.filter((variant) => variant.guidePath).map((variant) => ( + + ))} + +
+ ) : null} + + {offering.length ? ( +
    + {offering.map(({ provider: entry, options }) => { + const key = rowKey(entry.id, capability); + const chosen = chosenFor(entry, options); + const install = installLabel(chosen.capability, generatable, t); + const installedHere = (detection?.installed ?? []).find((item) => item.providerId === entry.id && item.capability === capability && item.model === chosen.model); + const state = probes[key]; + const verdict = state?.result && state.result.model === chosen.model ? describeProbe(state.result, t) : null; + const selectID = `multimodal-${entry.id}-${capability}`; + return ( +
  • +
    + {entry.name || entry.id} + {statusProviders?.[entry.id]?.custom ? {t("用户添加")} : null} +
    +
    + {options.length > 1 ? ( + <> + + + + ) : {chosen.model}} + {sourceLabel(chosen.capability, t)} + {installedHere + ? {t("已安装到 {agents}", { agents: installedHere.agents.map(nameOf).join(separator) })} + : {install.text}} + + {chosen.capability.probe ? ( + probe({ key, provider: entry, capability, model: chosen.model, adapter: chosen.capability.adapter, kind: chosen.capability.probe })} + /> + ) : null} + {install.ready ? ( + + ) : null} + +
    + {verdict ? : null} + {state?.error && state.model === chosen.model ?

    {state.error}

    : null} +
  • + ); + })} +
+ ) : ( +

{t("还没有已保存 API Key 的模型服务能做{capability}。", { capability: capabilityName })}

+ )} + + {needsFFmpeg ? ( +

{t("视频理解要先用 ffmpeg 抽帧。BootAgent 没有在 PATH 上找到 ffmpeg;请在 Agent 运行的环境里安装它。")}

+ ) : null} + {offering.length && without.length ? ( +

{t("{providers} 的模型列表里没有能做{capability}的模型。", { providers: without.map((entry) => entry.name || entry.id).join(separator), capability: capabilityName })}

+ ) : null} + {unsupported.length ? ( +

{t("{providers}:该模型服务目前没有可核实的公开接口,暂不支持{capability}。", { providers: unsupported.map((entry) => entry.name || entry.id).join(separator), capability: capabilityName })}

+ ) : null} + + {manualRoute && manualProvider ? ( +
+ {t("手动填写模型 ID")} +

{example + ? t("如果你知道能做{capability}的模型 ID(例如 {example}),可以填写后实测,通过后即可安装。", { capability: capabilityName, example }) + : t("如果你知道能做{capability}的模型 ID,可以填写后实测,通过后即可安装。", { capability: capabilityName })}

+
+ {accounts.length > 1 ? ( + + ) : null} + setManual((all) => ({ ...all, [capability]: { provider: manualProvider.id, model: event.target.value } }))} /> + probe({ key: manualKey(capability), provider: manualProvider, capability, model: (manualState?.model ?? "").trim(), adapter: manualRoute.adapter, kind: manualRoute.probe })} + /> +
+ {manualProbe?.result ? : null} + {manualProbe?.error ?

{manualProbe.error}

: null} +
+ ) : null} + + ); + }; + + const unlisted = accounts.filter((entry) => !entry.listed); return (

-

{t("用已保存 API Key 的模型服务查找能处理图片、音频和视频的模型。列出模型不收费;点击实测会发出真实请求。")}

+

{t("按能力查看已保存 API Key 的模型服务能做什么。列出模型不收费;点击实测会发出真实请求。")}

{error ?

{error}

: null} + {unlisted.map((entry) => ( +

{t("{provider}:无法读取模型列表:{message}", { provider: entry.name || entry.id, message: entry.message || entry.errorCode || "" })}

+ ))} - {installedCards.length ? ( -
-

{t("已安装的多模态 Skill")}

- {installedCards.map((card) => ( -
-
- {t(CAPABILITY_LABELS[card.capability])} - {card.skillId} - -
- {card.variants.map((variant) => ( -

- {variant.agents.length - ? t("{model}({provider})已安装到 {agents}", { model: variant.model, provider: providerName(variant.providerId), agents: variant.agents.map(nameOf).join(separator) }) - : t("{model}({provider})", { model: variant.model, provider: providerName(variant.providerId) })} -

- ))} - {card.variants.filter((variant) => variant.guidePath).map((variant) => ( - - ))} - -
- ))} -
- ) : null} - - {detection && !providers.length ? ( + {detection && !accounts.length ? (

{t("还没有保存 API Key 的模型服务。")} {t("去添加模型服务")}

) : null} - {providers.map((entry) => { - const groups = groupsByProvider.get(entry.id) ?? []; - const manualState = manual[entry.id]; - const manualRoutes = detection?.manual ?? []; - const manualRoute = manualRoutes.find((route) => route.capability === (manualState?.capability ?? manualRoutes[0]?.capability)); - const manualProbeState = manualRoute ? probes[rowKey(entry.id, manualRoute.capability)] : undefined; - const manualGroup = groups.find((group) => group.id === manualRoute?.capability); - const manualChosen = manualGroup?.options.find((option) => option.model === selected[rowKey(entry.id, manualRoute?.capability ?? "")]) ?? manualGroup?.options[0]; - const manualResultInRow = manualChosen?.model === manualProbeState?.model; - return ( -
-
- {entry.name || entry.id} - {!entry.listed ? {t("无法读取模型列表:{message}", { message: entry.message || entry.errorCode || "" })} : null} -
- {entry.unsupported.length ? ( -

{t("{capabilities}:该模型服务目前没有可核实的公开接口,暂不支持。", { capabilities: entry.unsupported.map((id) => t(CAPABILITY_LABELS[id])).join(separator) })}

- ) : null} - {!groups.length ? ( -

{entry.listed ? t("没有找到多模态模型。") : t("可以稍后重新检测。")}

- ) : ( -
    - {groups.map((group) => { - const key = rowKey(entry.id, group.id); - const chosen = group.options.find((option) => option.model === selected[key]) ?? group.options[0]; - const install = installLabel(chosen.capability, generatable, t); - const installedHere = (detection?.installed ?? []).find((item) => item.providerId === entry.id && item.capability === group.id && item.model === chosen.model); - const state = probes[key]; - const verdict = state?.result && state.result.model === chosen.model ? describeProbe(state.result, t) : null; - const selectID = `multimodal-${entry.id}-${group.id}`; - const capabilityName = t(CAPABILITY_LABELS[group.id]); - return ( -
  • - - - {sourceLabel(chosen.capability, t)} - {installedHere - ? {t("已安装到 {agents}", { agents: installedHere.agents.map(nameOf).join(separator) })} - : {install.text}} - {chosen.capability.probe ? ( - probe({ provider: entry, capability: group.id, model: chosen.model, adapter: chosen.capability.adapter, kind: chosen.capability.probe })} - /> - ) : null} - {install.ready ? ( - - ) : null} - {chosen.capability.adapter === "frames-chat-vision" && detection && !detection.ffmpeg ? ( -

    {t("视频理解要先用 ffmpeg 抽帧。BootAgent 没有在 PATH 上找到 ffmpeg;请在 Agent 运行的环境里安装它。")}

    - ) : null} - {verdict ? : null} - {state?.error && state.model === chosen.model ?

    {state.error}

    : null} -
  • - ); - })} -
- )} - {manualRoutes.length ? ( -
- {t("手动填写模型 ID")} -

{t("模型列表里没有专用的图片生成、语音识别或语音合成模型。如果你知道模型 ID(例如 gpt-image-1、whisper-1、tts-1),可以填写后实测,通过后即可安装。")}

-
- - setManual((current) => ({ ...current, [entry.id]: { capability: manualRoute?.capability ?? "vision", model: event.target.value } }))} /> - {manualRoute ? ( - probe({ provider: entry, capability: manualRoute.capability, model: (manualState?.model ?? "").trim(), adapter: manualRoute.adapter, kind: manualRoute.probe })} - /> - ) : null} -
- {manualProbeState?.result && !manualResultInRow - ? - : null} - {manualProbeState?.error && !manualResultInRow ?

{manualProbeState.error}

: null} -
- ) : null} + + {detection && (accounts.length || installedCards.length) ? ( + <> +
+ {CAPABILITY_ORDER.map((id, index) => { + const count = byCapability.get(id)?.length ?? 0; + const installed = installedCards.some((card) => card.capability === id); + return ( + + ); + })}
- ); - })} +
+ {renderPanel(current)} +
+ + ) : null} {confirmProbe ? ( setConfirmProbe(null)}> diff --git a/frontend/src/i18n.tsx b/frontend/src/i18n.tsx index 98609325..10cdc9eb 100644 --- a/frontend/src/i18n.tsx +++ b/frontend/src/i18n.tsx @@ -917,7 +917,16 @@ const english = { "下载量": "Downloads", // multimodal capabilities "多模态能力": "Multimodal capabilities", - "用已保存 API Key 的模型服务查找能处理图片、音频和视频的模型。列出模型不收费;点击实测会发出真实请求。": "Finds models that handle images, audio and video on the providers you have saved an API key for. Listing models is free; Probe sends a real request.", + "按能力查看已保存 API Key 的模型服务能做什么。列出模型不收费;点击实测会发出真实请求。": "See what the providers you have saved an API key for can do, one capability at a time. Listing models is free; Probe sends a real request.", + "多模态能力分类": "Multimodal capabilities", + "{count} 个账号可用": "{count} accounts available", + "{provider} 的{capability}模型": "{provider} {capability} model", + "{provider}:无法读取模型列表:{message}": "{provider}: could not read the model list: {message}", + "还没有已保存 API Key 的模型服务能做{capability}。": "None of the providers you have saved an API key for can do {capability} yet.", + "{providers} 的模型列表里没有能做{capability}的模型。": "The model list of {providers} has no model for {capability}.", + "{providers}:该模型服务目前没有可核实的公开接口,暂不支持{capability}。": "{providers}: no public API for {capability} is documented yet, so it is not supported.", + "如果你知道能做{capability}的模型 ID(例如 {example}),可以填写后实测,通过后即可安装。": "If you know a model ID for {capability} (such as {example}), enter it and probe it; once it passes it can be installed.", + "如果你知道能做{capability}的模型 ID,可以填写后实测,通过后即可安装。": "If you know a model ID for {capability}, enter it and probe it; once it passes it can be installed.", "图片理解": "Image understanding", "图片生成": "Image generation", "语音识别": "Speech-to-text", @@ -942,9 +951,6 @@ const english = { "无法检测多模态能力": "Could not detect multimodal capabilities", "还没有保存 API Key 的模型服务。": "No provider has a saved API key yet.", "去添加模型服务": "Add a provider", - "无法读取模型列表:{message}": "Could not read the model list: {message}", - "没有找到多模态模型。": "No multimodal models found.", - "可以稍后重新检测。": "You can detect again later.", "向模型服务发送一次带小图片的真实请求,确认这个模型能做{capability}。可能产生少量费用。": "Sends one real request with a small image to confirm this model can do {capability}. This may incur a small charge.", "正在实测": "Probing", "实测": "Probe", @@ -965,14 +971,11 @@ const english = { "将从 {agents} 移除这个 Skill。BootAgent 会保留一份备份;没有其他 Skill 再用到的 API Key 环境变量也会一并移除。": "This removes the Skill from {agents}. BootAgent keeps a backup, and removes any API key variable no other Skill still needs.", "将替换这些 Agent 当前使用的 {models}。": "This replaces {models} on these Agents.", "已安装到 {agents}": "Installed in {agents}", - "已安装的多模态 Skill": "Installed multimodal Skills", "手动填写模型 ID": "Enter a model ID", "暂不支持安装": "Install not supported yet", "模型 ID": "Model ID", - "模型列表里没有专用的图片生成、语音识别或语音合成模型。如果你知道模型 ID(例如 gpt-image-1、whisper-1、tts-1),可以填写后实测,通过后即可安装。": "Model lists do not include dedicated image generation, speech-to-text or text-to-speech models. If you know a model ID (gpt-image-1, whisper-1, tts-1), enter it and probe it; once it passes it can be installed.", "正在卸载": "Uninstalling", "正在运行的 Agent 需要重启,才能读到新写入的环境变量。": "Restart any Agent that is already running so it picks up the new variable.", - "没有找到可以安装 Skill 的 Agent": "No Agent that can take this Skill was found", "用「{model}」({provider})为 Agent 提供{capability}。": "Give Agents {capability} with \"{model}\" ({provider}).", "确认实测": "Confirm the probe", "这次实测会让「{model}」真实生成一张图片,按模型服务的价格计费。确定要发送吗?": "This probe makes \"{model}\" generate one real image, billed at the provider's price. Send it?", @@ -983,7 +986,6 @@ const english = { "其他 Agent(说明文件)": "Other Agents (guide)", "其他 Agent(生成说明文件)": "Other Agents (write a guide)", "{agents}没有可配置的环境变量,Key 会写进 BootAgent 的私有文件 {path},由 Skill 的脚本读取。": "{agents} have no configurable environment, so the key goes into BootAgent's private file {path}, which the Skill's script reads.", - "{capabilities}:该模型服务目前没有可核实的公开接口,暂不支持。": "{capabilities}: this provider documents no public API for it yet, so it is not supported.", } as const; export type Locale = "zh-CN" | "en"; diff --git a/frontend/src/pages/MultimodalPage.tsx b/frontend/src/pages/MultimodalPage.tsx index 14157916..077f02bf 100644 --- a/frontend/src/pages/MultimodalPage.tsx +++ b/frontend/src/pages/MultimodalPage.tsx @@ -8,7 +8,8 @@ import { useWizard } from "../state/WizardContext"; /** * Multimodal capabilities get their own page: below a long Skill library the * section was out of sight. The Skills it installs still show up, and can be - * retargeted, on the Skills page. + * retargeted, on the Skills page. The status's Provider records order the + * accounts within each capability, the ones the user added first. */ export function MultimodalPage() { const { t } = useI18n(); @@ -19,7 +20,7 @@ export function MultimodalPage() { ); return ( - + ); } diff --git a/frontend/src/styles/app.css b/frontend/src/styles/app.css index b6aced90..25ad7678 100644 --- a/frontend/src/styles/app.css +++ b/frontend/src/styles/app.css @@ -3944,21 +3944,22 @@ .marketplace-grid { grid-template-columns: repeat(2, minmax(0, 1fr)); grid-auto-rows: 108px; } } -/* Multimodal capabilities on the Skills page: one block per keyed Provider, - one row per capability. */ +/* Multimodal capabilities page: one tab per capability, one row per account + (a Provider with a saved key) that can provide it. */ .multimodal-section .section-heading h2 { display: flex; align-items: center; gap: 6px; } -.multimodal-provider { display: grid; gap: 8px; padding: 12px 14px; border: 1px solid var(--border); border-radius: var(--radius-panel); } -.multimodal-provider > header { display: flex; flex-wrap: wrap; align-items: baseline; gap: 10px; font-size: 13px; } -.multimodal-provider-note { color: var(--text-secondary); font-size: 12px; overflow-wrap: anywhere; } -.multimodal-capabilities { margin: 0; padding: 0; list-style: none; display: grid; gap: 8px; } -.multimodal-capabilities > li { display: flex; flex-wrap: wrap; align-items: center; gap: 8px; font-size: 13px; } -.multimodal-capability-name { min-width: 88px; font-weight: 600; } -.multimodal-capabilities select { min-width: 0; max-width: 320px; min-height: 28px; padding: 0 8px; border: 1px solid var(--border); border-radius: var(--radius-control); background: var(--window-bg); color: var(--text-primary); font: inherit; font-size: 12px; } -.multimodal-verdict { flex-basis: 100%; margin: 0; } +.multimodal-tabs { margin-bottom: 0; } +.multimodal-tab-installed { display: inline-flex; color: var(--green); } +.multimodal-panel { display: grid; gap: 10px; } +.multimodal-accounts { margin: 0; padding: 0; list-style: none; display: grid; gap: 8px; } +.multimodal-account { display: grid; gap: 8px; padding: 12px 14px; border: 1px solid var(--border); border-radius: var(--radius-panel); font-size: 13px; } +.multimodal-account-row { display: flex; flex-wrap: wrap; align-items: center; gap: 8px; } +.multimodal-account-actions { display: inline-flex; flex-wrap: wrap; gap: 8px; margin-left: auto; } +.multimodal-model { font-size: 12px; overflow-wrap: anywhere; } +.multimodal-provider-note { margin: 0; color: var(--text-secondary); font-size: 12px; overflow-wrap: anywhere; } +.multimodal-account select { min-width: 0; max-width: 320px; min-height: 28px; padding: 0 8px; border: 1px solid var(--border); border-radius: var(--radius-control); background: var(--window-bg); color: var(--text-primary); font: inherit; font-size: 12px; } +.multimodal-verdict { margin: 0; } .multimodal-empty { margin: 0; color: var(--text-secondary); font-size: 13px; } .multimodal-error { margin: 0; color: var(--red); font-size: 13px; } -.multimodal-installed { display: grid; gap: 8px; } -.multimodal-installed h3 { margin: 0; font-size: 13px; font-weight: 650; } .multimodal-installed-card { display: grid; gap: 6px; padding: 12px 14px; border: 1px solid var(--border); border-radius: var(--radius-panel); background: var(--surface-subtle); } .multimodal-installed-card > header { display: flex; align-items: center; gap: 10px; font-size: 13px; } .multimodal-installed-card > header code { color: var(--text-secondary); font-size: 12px; }