diff --git a/frontend/src/components/MultimodalSection.test.tsx b/frontend/src/components/MultimodalSection.test.tsx index 0be2e3c5..4d325043 100644 --- a/frontend/src/components/MultimodalSection.test.tsx +++ b/frontend/src/components/MultimodalSection.test.tsx @@ -2,7 +2,7 @@ import { act, fireEvent, render, screen, waitFor, within } from "@testing-librar import { MemoryRouter } from "react-router-dom"; import { beforeEach, describe, expect, it, vi } from "vitest"; -import type { MultimodalDetection, MultimodalProvider } from "../types/api"; +import type { MultimodalDetection, MultimodalProvider, StatusResponse } from "../types/api"; const detectMultimodal = vi.fn<() => Promise>(); const probeMultimodal = vi.fn(); @@ -23,7 +23,7 @@ vi.mock("../backend/api", () => ({ vi.mock("../utils/clipboard", () => ({ copyToClipboard: (text: string) => copyToClipboard(text) })); import { I18nProvider, sourceTranslate } from "../i18n"; -import { describeProbe, groupByCapability, groupInstalled, installLabel, MultimodalSection } from "./MultimodalSection"; +import { accountsFor, describeProbe, groupByCapability, groupInstalled, installLabel, MultimodalSection, orderAccounts } from "./MultimodalSection"; import { EXAMPLE_PROMPTS } from "./multimodalPrompts"; const GENERATABLE = ["openai-chat-vision", "openai-images", "openai-audio-transcriptions", "openai-audio-speech", "frames-chat-vision"]; @@ -68,11 +68,15 @@ function detection(partial: Partial = {}): MultimodalDetect const labels = { "claude-code": "Claude Code", codex: "Codex" }; -function renderSection(onSkillsChanged = vi.fn()) { - render(); +function renderSection(onSkillsChanged = vi.fn(), providers?: StatusResponse["providers"]) { + render(); return onSkillsChanged; } +const builtIn = (name: string, order: number) => ({ name, home: "https://example.com", base_url: "https://example.com/v1", order, has_key: true }); +const custom = (name: string, created: string) => ({ name, home: "", base_url: "https://gateway.example/v1", custom: true, has_key: true, created_at: created }); +const visionOnly = (id: string, name: string): MultimodalProvider => ({ ...gateway, id, name, models: [gateway.models[1]] }); + beforeEach(() => { for (const mock of [detectMultimodal, probeMultimodal, installMultimodal, uninstallSkill, copyToClipboard]) mock.mockClear(); detectMultimodal.mockReset(); @@ -87,6 +91,33 @@ describe("groupByCapability", () => { }); }); +describe("orderAccounts", () => { + it("lists the accounts the user added first, newest first, then built-ins in catalog order", () => { + const ordered = orderAccounts( + [visionOnly("ppio", "PPIO"), visionOnly("novita", "Novita"), visionOnly("old", "Old gateway"), visionOnly("new", "New gateway"), visionOnly("unknown", "Unknown")], + { novita: builtIn("Novita", 3), ppio: builtIn("PPIO", 2), old: custom("Old gateway", "2026-01-01T00:00:00Z"), new: custom("New gateway", "2026-09-01T00:00:00Z") }, + ); + expect(ordered.map((entry) => entry.id)).toEqual(["new", "old", "ppio", "novita", "unknown"]); + }); + + it("leaves out BootAgent's protocol converters", () => { + const ordered = orderAccounts( + [visionOnly("bootagent-converter-chat", "BootAgent Converter chat"), visionOnly("ppio", "PPIO")], + { "bootagent-converter-chat": custom("BootAgent Converter chat", "2026-10-01T00:00:00Z"), ppio: builtIn("PPIO", 2) }, + ); + expect(ordered.map((entry) => entry.id)).toEqual(["ppio"]); + }); + + it("keeps the backend order without status records", () => { + expect(orderAccounts([visionOnly("ppio", "PPIO"), visionOnly("mine", "Mine")]).map((entry) => entry.id)).toEqual(["ppio", "mine"]); + }); + + it("only lists accounts with a model for the capability", () => { + expect(accountsFor([gateway, { ...gateway, id: "empty", models: [] }], "vision").map((account) => account.provider.id)).toEqual(["novita"]); + expect(accountsFor([gateway], "image-generation")).toEqual([]); + }); +}); + describe("recommended models", () => { it("preselects the catalog's recommended model among equally usable ones", () => { const speech = (id: string, recommended: boolean) => ({ id, capabilities: [{ id: "text-to-speech" as const, source: "catalog" as const, adapter: "novita-audio", supported: true, installable: true, probe: "minimal" as const, recommended }] }); @@ -164,7 +195,7 @@ describe("MultimodalSection", () => { detectMultimodal.mockResolvedValue(detection()); probeMultimodal.mockResolvedValue({ providerId: "novita", model: "moonshotai/kimi-k3", capability: "vision", adapter: "openai-chat-vision", status: "available" }); renderSection(); - fireEvent.change(await screen.findByLabelText("图片理解"), { target: { value: "moonshotai/kimi-k3" } }); + fireEvent.change(await screen.findByLabelText("Novita 的图片理解模型"), { target: { value: "moonshotai/kimi-k3" } }); expect(screen.getByText("实测通过后可安装")).toBeTruthy(); // No install button while the evidence is only a guess. expect(screen.queryByRole("button", { name: /^安装$/ })).toBeNull(); @@ -178,8 +209,10 @@ describe("MultimodalSection", () => { detectMultimodal.mockResolvedValue(detection()); probeMultimodal.mockResolvedValue({ providerId: "novita", model: "gpt-image-1", capability: "image-generation", adapter: "openai-images", status: "available" }); renderSection(); - await screen.findByText("Novita"); - fireEvent.change(screen.getByLabelText("能力"), { target: { value: "image-generation" } }); + fireEvent.click(await screen.findByRole("tab", { name: /^图片生成/ })); + expect(screen.getByText("还没有已保存 API Key 的模型服务能做图片生成。")).toBeTruthy(); + // One account: nothing to choose between. + expect(screen.queryByLabelText("模型服务")).toBeNull(); fireEvent.change(screen.getByLabelText("模型 ID"), { target: { value: "gpt-image-1" } }); const manualProbe = within(screen.getByText("手动填写模型 ID").closest("details") as HTMLElement).getByRole("button", { name: /^实测$/ }); expect(manualProbe.getAttribute("title")).toContain("发送前会先确认"); @@ -248,12 +281,13 @@ describe("MultimodalSection", () => { await waitFor(() => expect(installMultimodal).toHaveBeenCalledWith({ provider: "novita", model: "qwen/qwen3-vl-8b", capability: "vision", agents: [], guide: true })); }); - it.each(["network", "key", "transient"])("shows a manual model's %s result despite an existing capability row", async (reason) => { + it.each(["network", "key", "transient"])("shows a manual model's %s result beside an account row for the capability", async (reason) => { detectMultimodal.mockResolvedValue(detection({ providers: [{ ...gateway, models: [{ id: "known-whisper", capabilities: [{ id: "speech-to-text", source: "catalog", adapter: "openai-audio-transcriptions", supported: true, installable: true, probe: "minimal" }] }] }] })); probeMultimodal.mockResolvedValue({ providerId: "novita", model: "my-whisper", capability: "speech-to-text", adapter: "openai-audio-transcriptions", status: reason === "key" ? "unavailable" : "unverified", reason }); renderSection(); - await screen.findByText("Novita"); - fireEvent.change(screen.getByLabelText("能力"), { target: { value: "speech-to-text" } }); + // The only capability on offer is the one opened by default. + expect(await screen.findByRole("tab", { name: /^语音识别/, selected: true })).toBeTruthy(); + expect(screen.getByText("known-whisper")).toBeTruthy(); fireEvent.change(screen.getByLabelText("模型 ID"), { target: { value: "my-whisper" } }); const manual = screen.getByText("手动填写模型 ID").closest("details") as HTMLElement; fireEvent.click(within(manual).getByRole("button", { name: /^实测$/ })); @@ -264,8 +298,7 @@ describe("MultimodalSection", () => { detectMultimodal.mockResolvedValue(detection({ providers: [{ ...gateway, models: [] }] })); probeMultimodal.mockRejectedValue(new Error("manual request failed")); renderSection(); - await screen.findByText("Novita"); - fireEvent.change(screen.getByLabelText("能力"), { target: { value: "speech-to-text" } }); + fireEvent.click(await screen.findByRole("tab", { name: /^语音识别/ })); fireEvent.change(screen.getByLabelText("模型 ID"), { target: { value: "my-whisper" } }); const manual = screen.getByText("手动填写模型 ID").closest("details") as HTMLElement; fireEvent.click(within(manual).getByRole("button", { name: /^实测$/ })); @@ -303,7 +336,8 @@ describe("MultimodalSection", () => { it("says which capabilities a Provider documents no API for", async () => { detectMultimodal.mockResolvedValue(detection({ providers: [{ ...gateway, unsupported: ["video-generation"] }] })); renderSection(); - expect(await screen.findByText("视频生成:该模型服务目前没有可核实的公开接口,暂不支持。")).toBeTruthy(); + fireEvent.click(await screen.findByRole("tab", { name: /^视频生成/ })); + expect(screen.getByText("Novita:该模型服务目前没有可核实的公开接口,暂不支持视频生成。")).toBeTruthy(); }); it("offers install without a probe for a catalog video model", async () => { @@ -312,9 +346,9 @@ describe("MultimodalSection", () => { providers: [{ ...gateway, models: [{ id: "wan2.7-t2v", capabilities: [{ id: "video-generation", source: "catalog", adapter: "novita-async", supported: true, installable: true, probe: "", recommended: true }] }] }], })); renderSection(); - expect(await screen.findByLabelText("视频生成")).toBeTruthy(); + expect(await screen.findByRole("tab", { name: /^视频生成/, selected: true })).toBeTruthy(); expect(screen.getByText("内置目录")).toBeTruthy(); - const row = screen.getByLabelText("视频生成").closest("li") as HTMLElement; + const row = screen.getByText("wan2.7-t2v").closest("li") as HTMLElement; expect(within(row).queryByRole("button", { name: /^实测$/ })).toBeNull(); expect(within(row).getByRole("button", { name: /^安装$/ })).toBeTruthy(); }); @@ -337,10 +371,11 @@ describe("MultimodalSection", () => { expect(copyToClipboard).toHaveBeenCalledWith(expect.stringContaining("先读 /Users/me/.bootagent/multimodal/vision/SKILL.md")); }); - it("warns when ffmpeg is missing for video by frames", async () => { - detectMultimodal.mockResolvedValue(detection({ ffmpeg: false, providers: [{ ...gateway, models: [{ id: "qwen/qwen3-vl-8b", capabilities: [{ id: "video-understanding", source: "declared", adapter: "frames-chat-vision", supported: true, installable: true, probe: "minimal" }] }] }] })); + it("warns once per tab when ffmpeg is missing for video by frames", async () => { + const frames: MultimodalProvider = { ...gateway, models: [{ id: "qwen/qwen3-vl-8b", capabilities: [{ id: "video-understanding", source: "declared", adapter: "frames-chat-vision", supported: true, installable: true, probe: "minimal" }] }] }; + detectMultimodal.mockResolvedValue(detection({ ffmpeg: false, providers: [frames, { ...frames, id: "ppio", name: "PPIO" }] })); renderSection(); - expect(await screen.findByText(/没有在 PATH 上找到 ffmpeg/)).toBeTruthy(); + expect(await screen.findAllByText(/没有在 PATH 上找到 ffmpeg/)).toHaveLength(1); }); it("points to the Providers page when no key is saved", async () => { @@ -349,9 +384,67 @@ describe("MultimodalSection", () => { expect(await screen.findByText("去添加模型服务")).toBeTruthy(); }); + it("puts generation first and opens the first capability an account can provide", async () => { + detectMultimodal.mockResolvedValue(detection()); + renderSection(); + const tabs = await screen.findAllByRole("tab"); + expect(tabs.map((tab) => tab.textContent)).toEqual(["图片生成", "视频生成", "图片理解1", "视频理解1", "语音合成", "语音识别"]); + expect(screen.getByRole("tab", { name: /^图片理解/ })).toHaveAttribute("aria-selected", "true"); + expect(screen.getByRole("tabpanel")).toHaveAttribute("aria-labelledby", "multimodal-tab-vision"); + }); + + it("opens image generation by default when an account can do it", async () => { + const images = { id: "gpt-image-1", capabilities: [{ id: "image-generation" as const, source: "probe" as const, adapter: "openai-images", supported: true, installable: true, probe: "confirm" as const }] }; + detectMultimodal.mockResolvedValue(detection({ providers: [{ ...gateway, models: [...gateway.models, images] }] })); + renderSection(); + expect(await screen.findByRole("tab", { name: /^图片生成/, selected: true })).toBeTruthy(); + expect(screen.getByText("gpt-image-1")).toBeTruthy(); + expect(screen.queryByText("还没有已保存 API Key 的模型服务能做图片生成。")).toBeNull(); + }); + + it("switches capability by click and arrow keys", async () => { + detectMultimodal.mockResolvedValue(detection()); + renderSection(); + const vision = await screen.findByRole("tab", { name: /^图片理解/ }); + fireEvent.keyDown(vision, { key: "ArrowRight" }); + expect(screen.getByRole("tab", { name: /^视频理解/ })).toHaveAttribute("aria-selected", "true"); + expect(document.activeElement).toBe(screen.getByRole("tab", { name: /^视频理解/ })); + fireEvent.keyDown(document.activeElement as HTMLElement, { key: "Home" }); + expect(screen.getByRole("tab", { name: /^图片生成/ })).toHaveAttribute("aria-selected", "true"); + fireEvent.click(screen.getByRole("tab", { name: /^图片理解/ })); + expect(screen.getByText("qwen/qwen3-vl-8b", { selector: "option" })).toBeTruthy(); + }); + + it("lists the account the user added before built-in ones and marks it", async () => { + detectMultimodal.mockResolvedValue(detection({ providers: [visionOnly("ppio", "PPIO"), visionOnly("mine", "我的网关")] })); + renderSection(vi.fn(), { ppio: builtIn("PPIO", 2), mine: custom("我的网关", "2026-09-01T00:00:00Z") }); + await screen.findByRole("tab", { name: /^图片理解2/ }); + const accounts = screen.getByRole("tabpanel").querySelectorAll(".multimodal-account"); + expect([...accounts].map((item) => item.querySelector("strong")?.textContent)).toEqual(["我的网关", "PPIO"]); + expect(within(accounts[0] as HTMLElement).getByText("用户添加")).toBeTruthy(); + expect(within(accounts[1] as HTMLElement).queryByText("用户添加")).toBeNull(); + }); + + it("names the accounts that have no model for the open capability", async () => { + detectMultimodal.mockResolvedValue(detection({ providers: [gateway, { ...gateway, id: "ppio", name: "PPIO", models: [] }] })); + renderSection(); + expect(await screen.findByText("PPIO 的模型列表里没有能做图片理解的模型。")).toBeTruthy(); + // With two accounts, the manual entry asks which one to probe on. + fireEvent.click(screen.getByRole("tab", { name: /^图片生成/ })); + expect(screen.getByLabelText("模型服务")).toBeTruthy(); + }); + + it("marks the tab of an installed capability", async () => { + detectMultimodal.mockResolvedValue(detection({ installed: [ + { skillId: "bootagent-vision", capability: "vision", providerId: "novita", model: "qwen/qwen3-vl-8b", adapter: "openai-chat-vision", agents: ["codex"] }, + ] })); + renderSection(); + expect(await screen.findByRole("tab", { name: /^图片理解已安装/ })).toBeTruthy(); + }); + it("reports a failed listing without hiding the Provider", async () => { detectMultimodal.mockResolvedValue(detection({ providers: [{ id: "deepseek", name: "DeepSeek", keyEnv: "BOOTAGENT_DEEPSEEK_API_KEY", keyFile: "/Users/me/.bootagent/skill-env/deepseek.env", unsupported: [], listed: false, message: "API key was rejected (401).", errorCode: "API_KEY_REJECTED", models: [] }] })); renderSection(); - expect(await screen.findByText(/无法读取模型列表/)).toBeTruthy(); + expect(await screen.findByText("DeepSeek:无法读取模型列表:API key was rejected (401).")).toBeTruthy(); }); }); diff --git a/frontend/src/components/MultimodalSection.tsx b/frontend/src/components/MultimodalSection.tsx index 6173e4e7..899b5862 100644 --- a/frontend/src/components/MultimodalSection.tsx +++ b/frontend/src/components/MultimodalSection.tsx @@ -1,10 +1,12 @@ import { Activity, Check, Copy, Download, Images, RefreshCw, Trash2, X } from "lucide-react"; -import { useCallback, useEffect, useMemo, useState } from "react"; +import { useCallback, useEffect, useMemo, useState, type KeyboardEvent } from "react"; import { Link } from "react-router-dom"; import { api, describeError } from "../backend/api"; import type { Translate, TranslationKey } from "../i18n"; import { useI18n } from "../i18n"; +import { isConverterID } from "../state/conversion"; +import { byProviderCreatedAt } from "../state/ranking"; import type { InstalledMultimodal, MultimodalCapability, @@ -15,18 +17,21 @@ import type { MultimodalProbeKind, MultimodalProvider, MultimodalSource, + StatusResponse, } from "../types/api"; import { copyToClipboard } from "../utils/clipboard"; import { ModalDialog } from "./ModalDialog"; import { EXAMPLE_PROMPTS } from "./multimodalPrompts"; -const CAPABILITY_ORDER: MultimodalCapabilityId[] = [ - "vision", +// Also the tab order. Generation leads: it is what users come to this page +// for, while understanding is something most chat models already do. +export const CAPABILITY_ORDER: MultimodalCapabilityId[] = [ "image-generation", - "speech-to-text", - "text-to-speech", "video-generation", + "vision", "video-understanding", + "text-to-speech", + "speech-to-text", ]; export const CAPABILITY_LABELS: Record = { @@ -38,6 +43,14 @@ export const CAPABILITY_LABELS: Record = "video-understanding": "视频理解", }; +// Shown in the manual entry hint: model IDs a user is likely to know. +const MANUAL_EXAMPLES: Partial> = { + vision: "gpt-4o-mini", + "image-generation": "gpt-image-1", + "speech-to-text": "whisper-1", + "text-to-speech": "tts-1", +}; + // Strongest evidence first. An ID rule is a guess from the name, so it sorts // last and is never preselected over anything the Provider or a probe said. const SOURCE_RANK: Record = { probe: 0, declared: 1, catalog: 2, "id-rule": 3 }; @@ -80,6 +93,31 @@ export function groupByCapability(entry: MultimodalProvider): CapabilityGroup[] })); } +/** + * Orders the keyed Providers the way the Providers page lists accounts: the + * ones the user added first, newest first, then built-ins in catalog order. + * The backend lists built-ins first; a Provider the status does not know keeps + * its place after the known ones. + */ +export function orderAccounts(providers: MultimodalProvider[], status?: StatusResponse["providers"]): MultimodalProvider[] { + const rank = new Map(byProviderCreatedAt(status ?? {}).map(([id], index) => [id, index])); + return providers + // BootAgent's own protocol converters hold a key but are not accounts; + // the Providers page hides them too. + .filter((entry) => !isConverterID(entry.id)) + .map((entry, index) => ({ entry, index })) + .sort((left, right) => (rank.get(left.entry.id) ?? rank.size + left.index) - (rank.get(right.entry.id) ?? rank.size + right.index)) + .map(({ entry }) => entry); +} + +/** One capability's accounts: every keyed Provider with a model for it, in account order. */ +export function accountsFor(providers: MultimodalProvider[], capability: MultimodalCapabilityId): Array<{ provider: MultimodalProvider; options: CapabilityOption[] }> { + return providers.flatMap((provider) => { + const group = groupByCapability(provider).find((item) => item.id === capability); + return group ? [{ provider, options: group.options }] : []; + }); +} + export function sourceLabel(capability: MultimodalCapability, t: Translate): string { switch (capability.source) { case "probe": @@ -159,36 +197,43 @@ export function groupInstalled(installed: InstalledMultimodal[]): Array<{ skillI .map(([skillId, variants]) => ({ skillId, capability, variants, agents: [...new Set(variants.flatMap((variant) => variant.agents))].sort() }))); } +// `key` is where the result is shown: an account's row, or a tab's manual entry. type ProbeState = { model: string; checking: boolean; result?: MultimodalProbe; error?: string }; -type ProbeTarget = { provider: MultimodalProvider; capability: MultimodalCapabilityId; model: string; adapter: string; kind: MultimodalProbeKind }; +type ProbeTarget = { key: string; provider: MultimodalProvider; capability: MultimodalCapabilityId; model: string; adapter: string; kind: MultimodalProbeKind }; type InstallTarget = { provider: MultimodalProvider; capability: MultimodalCapabilityId; model: string }; const rowKey = (provider: string, capability: string) => `${provider}::${capability}`; +const manualKey = (capability: string) => `manual::${capability}`; interface MultimodalSectionProps { /** Agent ID to display name. */ agentLabels?: Record; + /** The status's Provider records: account order and which ones the user added. */ + providers?: StatusResponse["providers"]; /** Called after an install or uninstall changed the Skill library. */ onSkillsChanged?: () => void; } /** - * The Skills page's Multimodal capabilities section. Opening it lists models, - * which is free; nothing billed is sent until the user presses a check button, - * and a check that produces billed output asks first. + * The Multimodal page. Each Provider with a saved key is an account; one tab + * per capability lists the accounts that can provide it, the ones the user + * added first. Opening it lists models, which is free; nothing billed is sent + * until the user presses a check button, and a check that produces billed + * output asks first. */ -export function MultimodalSection({ agentLabels = {}, onSkillsChanged }: MultimodalSectionProps) { +export function MultimodalSection({ agentLabels = {}, providers: statusProviders, onSkillsChanged }: MultimodalSectionProps) { const { t, locale } = useI18n(); const [detection, setDetection] = useState(null); const [loading, setLoading] = useState(false); const [error, setError] = useState(""); + const [active, setActive] = useState(null); const [selected, setSelected] = useState>({}); const [probes, setProbes] = useState>({}); const [confirmProbe, setConfirmProbe] = useState(null); const [installTarget, setInstallTarget] = useState(null); const [uninstallTarget, setUninstallTarget] = useState<{ skillId: string; capability: MultimodalCapabilityId; agents: string[] } | null>(null); const [uninstalling, setUninstalling] = useState(false); - const [manual, setManual] = useState>({}); + const [manual, setManual] = useState>>({}); const nameOf = (agentId: string) => agentLabels[agentId] ?? agentId; const separator = locale === "en" ? ", " : "、"; @@ -204,18 +249,17 @@ export function MultimodalSection({ agentLabels = {}, onSkillsChanged }: Multimo useEffect(() => { void detect(); }, [detect]); const runProbe = (target: ProbeTarget, confirmed: boolean) => { - const key = rowKey(target.provider.id, target.capability); - setProbes((current) => ({ ...current, [key]: { model: target.model, checking: true } })); + setProbes((current) => ({ ...current, [target.key]: { model: target.model, checking: true } })); void api.probeMultimodal(target.provider.id, target.model, target.capability, target.adapter, confirmed) .then((result) => { - setProbes((current) => ({ ...current, [key]: { model: target.model, checking: false, result } })); - // Show the probed model in its row; a conclusive verdict is cached by - // the backend and outranks every other source once re-read. - setSelected((current) => ({ ...current, [key]: target.model })); + setProbes((current) => ({ ...current, [target.key]: { model: target.model, checking: false, result } })); + // Show the probed model in its account's row; a conclusive verdict is + // cached by the backend and outranks every other source once re-read. + setSelected((current) => ({ ...current, [rowKey(target.provider.id, target.capability)]: target.model })); return detect(); }) .catch((reason: unknown) => { - setProbes((current) => ({ ...current, [key]: { model: target.model, checking: false, error: describeError(reason, t("无法向模型服务发送验证请求。")).message } })); + setProbes((current) => ({ ...current, [target.key]: { model: target.model, checking: false, error: describeError(reason, t("无法向模型服务发送验证请求。")).message } })); }); }; const probe = (target: ProbeTarget) => (target.kind === "confirm" ? setConfirmProbe(target) : runProbe(target, false)); @@ -234,18 +278,166 @@ export function MultimodalSection({ agentLabels = {}, onSkillsChanged }: Multimo .finally(() => setUninstalling(false)); }; - const providers = useMemo(() => detection?.providers ?? [], [detection]); - const groupsByProvider = useMemo(() => new Map(providers.map((entry) => [entry.id, groupByCapability(entry)])), [providers]); - const providerName = (id: string) => providers.find((entry) => entry.id === id)?.name || id; - const generatable = detection?.generatable ?? []; + const accounts = useMemo(() => orderAccounts(detection?.providers ?? [], statusProviders), [detection, statusProviders]); + const byCapability = useMemo(() => new Map(CAPABILITY_ORDER.map((id) => [id, accountsFor(accounts, id)])), [accounts]); const installedCards = useMemo(() => groupInstalled(detection?.installed ?? []), [detection]); + const providerName = (id: string) => accounts.find((entry) => entry.id === id)?.name || statusProviders?.[id]?.name || id; + const generatable = detection?.generatable ?? []; + // Until the user picks a tab: the first capability something can be done with. + const current = active + ?? CAPABILITY_ORDER.find((id) => byCapability.get(id)?.length || installedCards.some((card) => card.capability === id)) + ?? CAPABILITY_ORDER[0]; + + const onTabKey = (event: KeyboardEvent, index: number) => { + if (!["ArrowLeft", "ArrowRight", "Home", "End"].includes(event.key)) return; + event.preventDefault(); + const last = CAPABILITY_ORDER.length - 1; + const next = CAPABILITY_ORDER[event.key === "Home" ? 0 : event.key === "End" ? last : (index + (event.key === "ArrowRight" ? 1 : -1) + CAPABILITY_ORDER.length) % CAPABILITY_ORDER.length]; + setActive(next); + document.getElementById(`multimodal-tab-${next}`)?.focus(); + }; + const renderPanel = (capability: MultimodalCapabilityId) => { + const capabilityName = t(CAPABILITY_LABELS[capability]); + const offering = byCapability.get(capability) ?? []; + const card = installedCards.find((item) => item.capability === capability); + const unsupported = accounts.filter((entry) => entry.unsupported.includes(capability)); + const without = accounts.filter((entry) => entry.listed && !entry.unsupported.includes(capability) && !offering.some((account) => account.provider.id === entry.id)); + const manualRoute = (detection?.manual ?? []).find((route) => route.capability === capability); + const manualState = manual[capability]; + const manualProvider = accounts.find((entry) => entry.id === manualState?.provider) ?? accounts[0]; + const manualProbe = probes[manualKey(capability)]; + const example = MANUAL_EXAMPLES[capability]; + const chosenFor = (entry: MultimodalProvider, options: CapabilityOption[]) => options.find((option) => option.model === selected[rowKey(entry.id, capability)]) ?? options[0]; + // Said once for the tab, not on every account that samples frames. + const needsFFmpeg = Boolean(detection && !detection.ffmpeg) && offering.some(({ provider: entry, options }) => chosenFor(entry, options).capability.adapter === "frames-chat-vision"); + return ( + <> + {card ? ( +
+
+ {t("已安装")} + {card.skillId} + +
+ {card.variants.map((variant) => ( +

+ {variant.agents.length + ? t("{model}({provider})已安装到 {agents}", { model: variant.model, provider: providerName(variant.providerId), agents: variant.agents.map(nameOf).join(separator) }) + : t("{model}({provider})", { model: variant.model, provider: providerName(variant.providerId) })} +

+ ))} + {card.variants.filter((variant) => variant.guidePath).map((variant) => ( + + ))} + +
+ ) : null} + + {offering.length ? ( +
    + {offering.map(({ provider: entry, options }) => { + const key = rowKey(entry.id, capability); + const chosen = chosenFor(entry, options); + const install = installLabel(chosen.capability, generatable, t); + const installedHere = (detection?.installed ?? []).find((item) => item.providerId === entry.id && item.capability === capability && item.model === chosen.model); + const state = probes[key]; + const verdict = state?.result && state.result.model === chosen.model ? describeProbe(state.result, t) : null; + const selectID = `multimodal-${entry.id}-${capability}`; + return ( +
  • +
    + {entry.name || entry.id} + {statusProviders?.[entry.id]?.custom ? {t("用户添加")} : null} +
    +
    + {options.length > 1 ? ( + <> + + + + ) : {chosen.model}} + {sourceLabel(chosen.capability, t)} + {installedHere + ? {t("已安装到 {agents}", { agents: installedHere.agents.map(nameOf).join(separator) })} + : {install.text}} + + {chosen.capability.probe ? ( + probe({ key, provider: entry, capability, model: chosen.model, adapter: chosen.capability.adapter, kind: chosen.capability.probe })} + /> + ) : null} + {install.ready ? ( + + ) : null} + +
    + {verdict ? : null} + {state?.error && state.model === chosen.model ?

    {state.error}

    : null} +
  • + ); + })} +
+ ) : ( +

{t("还没有已保存 API Key 的模型服务能做{capability}。", { capability: capabilityName })}

+ )} + + {needsFFmpeg ? ( +

{t("视频理解要先用 ffmpeg 抽帧。BootAgent 没有在 PATH 上找到 ffmpeg;请在 Agent 运行的环境里安装它。")}

+ ) : null} + {offering.length && without.length ? ( +

{t("{providers} 的模型列表里没有能做{capability}的模型。", { providers: without.map((entry) => entry.name || entry.id).join(separator), capability: capabilityName })}

+ ) : null} + {unsupported.length ? ( +

{t("{providers}:该模型服务目前没有可核实的公开接口,暂不支持{capability}。", { providers: unsupported.map((entry) => entry.name || entry.id).join(separator), capability: capabilityName })}

+ ) : null} + + {manualRoute && manualProvider ? ( +
+ {t("手动填写模型 ID")} +

{example + ? t("如果你知道能做{capability}的模型 ID(例如 {example}),可以填写后实测,通过后即可安装。", { capability: capabilityName, example }) + : t("如果你知道能做{capability}的模型 ID,可以填写后实测,通过后即可安装。", { capability: capabilityName })}

+
+ {accounts.length > 1 ? ( + + ) : null} + setManual((all) => ({ ...all, [capability]: { provider: manualProvider.id, model: event.target.value } }))} /> + probe({ key: manualKey(capability), provider: manualProvider, capability, model: (manualState?.model ?? "").trim(), adapter: manualRoute.adapter, kind: manualRoute.probe })} + /> +
+ {manualProbe?.result ? : null} + {manualProbe?.error ?

{manualProbe.error}

: null} +
+ ) : null} + + ); + }; + + const unlisted = accounts.filter((entry) => !entry.listed); return (

-

{t("用已保存 API Key 的模型服务查找能处理图片、音频和视频的模型。列出模型不收费;点击实测会发出真实请求。")}

+

{t("按能力查看已保存 API Key 的模型服务能做什么。列出模型不收费;点击实测会发出真实请求。")}

{error ?

{error}

: null} + {unlisted.map((entry) => ( +

{t("{provider}:无法读取模型列表:{message}", { provider: entry.name || entry.id, message: entry.message || entry.errorCode || "" })}

+ ))} - {installedCards.length ? ( -
-

{t("已安装的多模态 Skill")}

- {installedCards.map((card) => ( -
-
- {t(CAPABILITY_LABELS[card.capability])} - {card.skillId} - -
- {card.variants.map((variant) => ( -

- {variant.agents.length - ? t("{model}({provider})已安装到 {agents}", { model: variant.model, provider: providerName(variant.providerId), agents: variant.agents.map(nameOf).join(separator) }) - : t("{model}({provider})", { model: variant.model, provider: providerName(variant.providerId) })} -

- ))} - {card.variants.filter((variant) => variant.guidePath).map((variant) => ( - - ))} - -
- ))} -
- ) : null} - - {detection && !providers.length ? ( + {detection && !accounts.length ? (

{t("还没有保存 API Key 的模型服务。")} {t("去添加模型服务")}

) : null} - {providers.map((entry) => { - const groups = groupsByProvider.get(entry.id) ?? []; - const manualState = manual[entry.id]; - const manualRoutes = detection?.manual ?? []; - const manualRoute = manualRoutes.find((route) => route.capability === (manualState?.capability ?? manualRoutes[0]?.capability)); - const manualProbeState = manualRoute ? probes[rowKey(entry.id, manualRoute.capability)] : undefined; - const manualGroup = groups.find((group) => group.id === manualRoute?.capability); - const manualChosen = manualGroup?.options.find((option) => option.model === selected[rowKey(entry.id, manualRoute?.capability ?? "")]) ?? manualGroup?.options[0]; - const manualResultInRow = manualChosen?.model === manualProbeState?.model; - return ( -
-
- {entry.name || entry.id} - {!entry.listed ? {t("无法读取模型列表:{message}", { message: entry.message || entry.errorCode || "" })} : null} -
- {entry.unsupported.length ? ( -

{t("{capabilities}:该模型服务目前没有可核实的公开接口,暂不支持。", { capabilities: entry.unsupported.map((id) => t(CAPABILITY_LABELS[id])).join(separator) })}

- ) : null} - {!groups.length ? ( -

{entry.listed ? t("没有找到多模态模型。") : t("可以稍后重新检测。")}

- ) : ( -
    - {groups.map((group) => { - const key = rowKey(entry.id, group.id); - const chosen = group.options.find((option) => option.model === selected[key]) ?? group.options[0]; - const install = installLabel(chosen.capability, generatable, t); - const installedHere = (detection?.installed ?? []).find((item) => item.providerId === entry.id && item.capability === group.id && item.model === chosen.model); - const state = probes[key]; - const verdict = state?.result && state.result.model === chosen.model ? describeProbe(state.result, t) : null; - const selectID = `multimodal-${entry.id}-${group.id}`; - const capabilityName = t(CAPABILITY_LABELS[group.id]); - return ( -
  • - - - {sourceLabel(chosen.capability, t)} - {installedHere - ? {t("已安装到 {agents}", { agents: installedHere.agents.map(nameOf).join(separator) })} - : {install.text}} - {chosen.capability.probe ? ( - probe({ provider: entry, capability: group.id, model: chosen.model, adapter: chosen.capability.adapter, kind: chosen.capability.probe })} - /> - ) : null} - {install.ready ? ( - - ) : null} - {chosen.capability.adapter === "frames-chat-vision" && detection && !detection.ffmpeg ? ( -

    {t("视频理解要先用 ffmpeg 抽帧。BootAgent 没有在 PATH 上找到 ffmpeg;请在 Agent 运行的环境里安装它。")}

    - ) : null} - {verdict ? : null} - {state?.error && state.model === chosen.model ?

    {state.error}

    : null} -
  • - ); - })} -
- )} - {manualRoutes.length ? ( -
- {t("手动填写模型 ID")} -

{t("模型列表里没有专用的图片生成、语音识别或语音合成模型。如果你知道模型 ID(例如 gpt-image-1、whisper-1、tts-1),可以填写后实测,通过后即可安装。")}

-
- - setManual((current) => ({ ...current, [entry.id]: { capability: manualRoute?.capability ?? "vision", model: event.target.value } }))} /> - {manualRoute ? ( - probe({ provider: entry, capability: manualRoute.capability, model: (manualState?.model ?? "").trim(), adapter: manualRoute.adapter, kind: manualRoute.probe })} - /> - ) : null} -
- {manualProbeState?.result && !manualResultInRow - ? - : null} - {manualProbeState?.error && !manualResultInRow ?

{manualProbeState.error}

: null} -
- ) : null} + + {detection && (accounts.length || installedCards.length) ? ( + <> +
+ {CAPABILITY_ORDER.map((id, index) => { + const count = byCapability.get(id)?.length ?? 0; + const installed = installedCards.some((card) => card.capability === id); + return ( + + ); + })}
- ); - })} +
+ {renderPanel(current)} +
+ + ) : null} {confirmProbe ? ( setConfirmProbe(null)}> diff --git a/frontend/src/i18n.tsx b/frontend/src/i18n.tsx index 98609325..10cdc9eb 100644 --- a/frontend/src/i18n.tsx +++ b/frontend/src/i18n.tsx @@ -917,7 +917,16 @@ const english = { "下载量": "Downloads", // multimodal capabilities "多模态能力": "Multimodal capabilities", - "用已保存 API Key 的模型服务查找能处理图片、音频和视频的模型。列出模型不收费;点击实测会发出真实请求。": "Finds models that handle images, audio and video on the providers you have saved an API key for. Listing models is free; Probe sends a real request.", + "按能力查看已保存 API Key 的模型服务能做什么。列出模型不收费;点击实测会发出真实请求。": "See what the providers you have saved an API key for can do, one capability at a time. Listing models is free; Probe sends a real request.", + "多模态能力分类": "Multimodal capabilities", + "{count} 个账号可用": "{count} accounts available", + "{provider} 的{capability}模型": "{provider} {capability} model", + "{provider}:无法读取模型列表:{message}": "{provider}: could not read the model list: {message}", + "还没有已保存 API Key 的模型服务能做{capability}。": "None of the providers you have saved an API key for can do {capability} yet.", + "{providers} 的模型列表里没有能做{capability}的模型。": "The model list of {providers} has no model for {capability}.", + "{providers}:该模型服务目前没有可核实的公开接口,暂不支持{capability}。": "{providers}: no public API for {capability} is documented yet, so it is not supported.", + "如果你知道能做{capability}的模型 ID(例如 {example}),可以填写后实测,通过后即可安装。": "If you know a model ID for {capability} (such as {example}), enter it and probe it; once it passes it can be installed.", + "如果你知道能做{capability}的模型 ID,可以填写后实测,通过后即可安装。": "If you know a model ID for {capability}, enter it and probe it; once it passes it can be installed.", "图片理解": "Image understanding", "图片生成": "Image generation", "语音识别": "Speech-to-text", @@ -942,9 +951,6 @@ const english = { "无法检测多模态能力": "Could not detect multimodal capabilities", "还没有保存 API Key 的模型服务。": "No provider has a saved API key yet.", "去添加模型服务": "Add a provider", - "无法读取模型列表:{message}": "Could not read the model list: {message}", - "没有找到多模态模型。": "No multimodal models found.", - "可以稍后重新检测。": "You can detect again later.", "向模型服务发送一次带小图片的真实请求,确认这个模型能做{capability}。可能产生少量费用。": "Sends one real request with a small image to confirm this model can do {capability}. This may incur a small charge.", "正在实测": "Probing", "实测": "Probe", @@ -965,14 +971,11 @@ const english = { "将从 {agents} 移除这个 Skill。BootAgent 会保留一份备份;没有其他 Skill 再用到的 API Key 环境变量也会一并移除。": "This removes the Skill from {agents}. BootAgent keeps a backup, and removes any API key variable no other Skill still needs.", "将替换这些 Agent 当前使用的 {models}。": "This replaces {models} on these Agents.", "已安装到 {agents}": "Installed in {agents}", - "已安装的多模态 Skill": "Installed multimodal Skills", "手动填写模型 ID": "Enter a model ID", "暂不支持安装": "Install not supported yet", "模型 ID": "Model ID", - "模型列表里没有专用的图片生成、语音识别或语音合成模型。如果你知道模型 ID(例如 gpt-image-1、whisper-1、tts-1),可以填写后实测,通过后即可安装。": "Model lists do not include dedicated image generation, speech-to-text or text-to-speech models. If you know a model ID (gpt-image-1, whisper-1, tts-1), enter it and probe it; once it passes it can be installed.", "正在卸载": "Uninstalling", "正在运行的 Agent 需要重启,才能读到新写入的环境变量。": "Restart any Agent that is already running so it picks up the new variable.", - "没有找到可以安装 Skill 的 Agent": "No Agent that can take this Skill was found", "用「{model}」({provider})为 Agent 提供{capability}。": "Give Agents {capability} with \"{model}\" ({provider}).", "确认实测": "Confirm the probe", "这次实测会让「{model}」真实生成一张图片,按模型服务的价格计费。确定要发送吗?": "This probe makes \"{model}\" generate one real image, billed at the provider's price. Send it?", @@ -983,7 +986,6 @@ const english = { "其他 Agent(说明文件)": "Other Agents (guide)", "其他 Agent(生成说明文件)": "Other Agents (write a guide)", "{agents}没有可配置的环境变量,Key 会写进 BootAgent 的私有文件 {path},由 Skill 的脚本读取。": "{agents} have no configurable environment, so the key goes into BootAgent's private file {path}, which the Skill's script reads.", - "{capabilities}:该模型服务目前没有可核实的公开接口,暂不支持。": "{capabilities}: this provider documents no public API for it yet, so it is not supported.", } as const; export type Locale = "zh-CN" | "en"; diff --git a/frontend/src/pages/MultimodalPage.tsx b/frontend/src/pages/MultimodalPage.tsx index 14157916..077f02bf 100644 --- a/frontend/src/pages/MultimodalPage.tsx +++ b/frontend/src/pages/MultimodalPage.tsx @@ -8,7 +8,8 @@ import { useWizard } from "../state/WizardContext"; /** * Multimodal capabilities get their own page: below a long Skill library the * section was out of sight. The Skills it installs still show up, and can be - * retargeted, on the Skills page. + * retargeted, on the Skills page. The status's Provider records order the + * accounts within each capability, the ones the user added first. */ export function MultimodalPage() { const { t } = useI18n(); @@ -19,7 +20,7 @@ export function MultimodalPage() { ); return ( - + ); } diff --git a/frontend/src/styles/app.css b/frontend/src/styles/app.css index b6aced90..25ad7678 100644 --- a/frontend/src/styles/app.css +++ b/frontend/src/styles/app.css @@ -3944,21 +3944,22 @@ .marketplace-grid { grid-template-columns: repeat(2, minmax(0, 1fr)); grid-auto-rows: 108px; } } -/* Multimodal capabilities on the Skills page: one block per keyed Provider, - one row per capability. */ +/* Multimodal capabilities page: one tab per capability, one row per account + (a Provider with a saved key) that can provide it. */ .multimodal-section .section-heading h2 { display: flex; align-items: center; gap: 6px; } -.multimodal-provider { display: grid; gap: 8px; padding: 12px 14px; border: 1px solid var(--border); border-radius: var(--radius-panel); } -.multimodal-provider > header { display: flex; flex-wrap: wrap; align-items: baseline; gap: 10px; font-size: 13px; } -.multimodal-provider-note { color: var(--text-secondary); font-size: 12px; overflow-wrap: anywhere; } -.multimodal-capabilities { margin: 0; padding: 0; list-style: none; display: grid; gap: 8px; } -.multimodal-capabilities > li { display: flex; flex-wrap: wrap; align-items: center; gap: 8px; font-size: 13px; } -.multimodal-capability-name { min-width: 88px; font-weight: 600; } -.multimodal-capabilities select { min-width: 0; max-width: 320px; min-height: 28px; padding: 0 8px; border: 1px solid var(--border); border-radius: var(--radius-control); background: var(--window-bg); color: var(--text-primary); font: inherit; font-size: 12px; } -.multimodal-verdict { flex-basis: 100%; margin: 0; } +.multimodal-tabs { margin-bottom: 0; } +.multimodal-tab-installed { display: inline-flex; color: var(--green); } +.multimodal-panel { display: grid; gap: 10px; } +.multimodal-accounts { margin: 0; padding: 0; list-style: none; display: grid; gap: 8px; } +.multimodal-account { display: grid; gap: 8px; padding: 12px 14px; border: 1px solid var(--border); border-radius: var(--radius-panel); font-size: 13px; } +.multimodal-account-row { display: flex; flex-wrap: wrap; align-items: center; gap: 8px; } +.multimodal-account-actions { display: inline-flex; flex-wrap: wrap; gap: 8px; margin-left: auto; } +.multimodal-model { font-size: 12px; overflow-wrap: anywhere; } +.multimodal-provider-note { margin: 0; color: var(--text-secondary); font-size: 12px; overflow-wrap: anywhere; } +.multimodal-account select { min-width: 0; max-width: 320px; min-height: 28px; padding: 0 8px; border: 1px solid var(--border); border-radius: var(--radius-control); background: var(--window-bg); color: var(--text-primary); font: inherit; font-size: 12px; } +.multimodal-verdict { margin: 0; } .multimodal-empty { margin: 0; color: var(--text-secondary); font-size: 13px; } .multimodal-error { margin: 0; color: var(--red); font-size: 13px; } -.multimodal-installed { display: grid; gap: 8px; } -.multimodal-installed h3 { margin: 0; font-size: 13px; font-weight: 650; } .multimodal-installed-card { display: grid; gap: 6px; padding: 12px 14px; border: 1px solid var(--border); border-radius: var(--radius-panel); background: var(--surface-subtle); } .multimodal-installed-card > header { display: flex; align-items: center; gap: 10px; font-size: 13px; } .multimodal-installed-card > header code { color: var(--text-secondary); font-size: 12px; }