diff --git a/.github/workflows/build-windows.yml b/.github/workflows/build-windows.yml new file mode 100644 index 0000000000..aee25dbe5a --- /dev/null +++ b/.github/workflows/build-windows.yml @@ -0,0 +1,62 @@ +name: Build Windows Native + +on: + push: + branches: + - feature/voice-input + workflow_dispatch: + +env: + RUST_BACKTRACE: 1 + HUSKY: 0 + +jobs: + build: + name: Build Windows App + runs-on: windows-latest + + steps: + - name: Checkout repository + uses: actions/checkout@v4 + + - name: Setup pnpm + uses: pnpm/action-setup@v4 + with: + version: 11.9.0 + + - name: Setup Node.js + uses: actions/setup-node@v4 + with: + node-version: 22 + cache: "pnpm" + + - name: Install Rust + uses: dtolnay/rust-toolchain@stable + with: + targets: x86_64-pc-windows-msvc + + - name: Rust Cache + uses: swatinem/rust-cache@v2 + with: + workspaces: "./src-tauri -> target" + + - name: Install dependencies + run: pnpm install --frozen-lockfile + + - name: Build codeg-server + run: pnpm server:build + + - name: Build Tauri App + run: pnpm tauri build --bundles nsis + env: + GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} + + - name: Upload Build Artifacts + uses: actions/upload-artifact@v4 + with: + name: codeg-windows-voice-build + path: | + src-tauri/target/release/codeg.exe + src-tauri/target/release/codeg-server.exe + src-tauri/target/release/bundle/nsis/*.exe + if-no-files-found: error diff --git a/src-tauri/tauri.conf.json b/src-tauri/tauri.conf.json index 57e249e45f..7ebbf016e6 100644 --- a/src-tauri/tauri.conf.json +++ b/src-tauri/tauri.conf.json @@ -18,7 +18,7 @@ }, "bundle": { "active": true, - "createUpdaterArtifacts": true, + "createUpdaterArtifacts": false, "targets": "all", "icon": [ "icons/32x32.png", diff --git a/src/components/chat/composer/composer-voice-button.test.tsx b/src/components/chat/composer/composer-voice-button.test.tsx new file mode 100644 index 0000000000..070c186fff --- /dev/null +++ b/src/components/chat/composer/composer-voice-button.test.tsx @@ -0,0 +1,77 @@ +import { describe, it, expect, vi } from "vitest" +import { render, screen, fireEvent } from "@testing-library/react" +import { ComposerVoiceButton } from "./composer-voice-button" +import type { UseVoiceInputReturn } from "./use-voice-input" + +function createMockVoice(overrides: Partial = {}): UseVoiceInputReturn { + return { + status: "idle", + mode: "web-speech", + isSupported: true, + volume: 0, + interimText: "", + errorMessage: null, + startListening: vi.fn(), + stopListening: vi.fn(), + cancel: vi.fn(), + ...overrides, + } +} + +describe("ComposerVoiceButton", () => { + it("renders idle button with microphone icon", () => { + const voice = createMockVoice() + render() + + const button = screen.getByTestId("composer-voice-button") + expect(button).toBeInTheDocument() + expect(button).not.toBeDisabled() + }) + + it("calls startListening on click when idle", () => { + const voice = createMockVoice() + render() + + fireEvent.click(screen.getByTestId("composer-voice-button")) + expect(voice.startListening).toHaveBeenCalled() + }) + + it("renders active state and calls stopListening on click when listening", () => { + const voice = createMockVoice({ status: "listening" }) + render() + + const button = screen.getByTestId("composer-voice-button") + expect(button.className).toContain("text-red-500") + + fireEvent.click(button) + expect(voice.stopListening).toHaveBeenCalled() + }) + + it("shows interim text bubble when speaking", () => { + const voice = createMockVoice({ + status: "listening", + interimText: "测试语音内容", + }) + render() + + const bubble = screen.getByTestId("voice-interim-bubble") + expect(bubble).toBeInTheDocument() + expect(bubble).toHaveTextContent("测试语音内容") + }) + + it("cancels on Escape key while listening", () => { + const voice = createMockVoice({ status: "listening" }) + render() + + fireEvent.keyDown(window, { key: "Escape" }) + expect(voice.cancel).toHaveBeenCalled() + }) + + it("disables button when transcribing", () => { + const voice = createMockVoice({ status: "transcribing" }) + render() + + const button = screen.getByTestId("composer-voice-button") + expect(button).toBeDisabled() + }) +}) diff --git a/src/components/chat/composer/composer-voice-button.tsx b/src/components/chat/composer/composer-voice-button.tsx new file mode 100644 index 0000000000..9000bc3f42 --- /dev/null +++ b/src/components/chat/composer/composer-voice-button.tsx @@ -0,0 +1,118 @@ +import { useEffect } from "react" +import { Button } from "@/components/ui/button" +import { Mic, Loader2 } from "lucide-react" +import { cn } from "@/lib/utils" +import type { UseVoiceInputReturn } from "./use-voice-input" +import { toast } from "sonner" + +export interface ComposerVoiceButtonProps { + voice: UseVoiceInputReturn + disabled?: boolean + className?: string + /** Localized string for default tooltip */ + label?: string + /** Localized string for listening state tooltip */ + listeningLabel?: string + /** Localized string for transcribing state tooltip */ + transcribingLabel?: string + /** Localized string for permission denied */ + permissionDeniedLabel?: string +} + +export function ComposerVoiceButton({ + voice, + disabled = false, + className, + label = "语音输入 (点击开始)", + listeningLabel = "正在录音... 点击完成 (Esc 取消)", + transcribingLabel = "正在转写...", + permissionDeniedLabel = "麦克风权限被拒绝,请在设置中允许访问", +}: ComposerVoiceButtonProps) { + const isListening = voice.status === "listening" + const isTranscribing = voice.status === "transcribing" + + // Cancel on Escape key while listening + useEffect(() => { + if (!isListening) return + + const handleKeyDown = (e: KeyboardEvent) => { + if (e.key === "Escape") { + e.preventDefault() + voice.cancel() + } + } + + window.addEventListener("keydown", handleKeyDown) + return () => window.removeEventListener("keydown", handleKeyDown) + }, [isListening, voice]) + + // Toast on permission error + useEffect(() => { + if (voice.status === "error" && voice.errorMessage === "micPermissionDenied") { + toast.error(permissionDeniedLabel) + } + }, [voice.status, voice.errorMessage, permissionDeniedLabel]) + + const handleClick = (e: React.MouseEvent) => { + e.preventDefault() + e.stopPropagation() + if (disabled || isTranscribing) return + + if (isListening) { + voice.stopListening() + } else { + void voice.startListening() + } + } + + const title = isTranscribing ? transcribingLabel : isListening ? listeningLabel : label + + return ( +
+ {/* Live speech interim preview bubble while user is speaking */} + {isListening && voice.interimText && ( +
+ + {voice.interimText} +
+ )} + + +
+ ) +} diff --git a/src/components/chat/composer/use-voice-input.test.ts b/src/components/chat/composer/use-voice-input.test.ts new file mode 100644 index 0000000000..57371ff6f5 --- /dev/null +++ b/src/components/chat/composer/use-voice-input.test.ts @@ -0,0 +1,142 @@ +import { describe, it, expect, vi, beforeEach, afterEach } from "vitest" +import { renderHook, act } from "@testing-library/react" +import { useVoiceInput } from "./use-voice-input" + +describe("useVoiceInput", () => { + let mockRecognitionInstance: any + + beforeEach(() => { + mockRecognitionInstance = { + start: vi.fn(), + stop: vi.fn(), + abort: vi.fn(), + continuous: false, + interimResults: false, + lang: "", + onresult: null, + onerror: null, + onend: null, + } + + ;(window as any).SpeechRecognition = vi.fn(() => mockRecognitionInstance) + }) + + afterEach(() => { + delete (window as any).SpeechRecognition + delete (window as any).webkitSpeechRecognition + vi.clearAllMocks() + }) + + it("initializes with idle status and web-speech support", () => { + const { result } = renderHook(() => useVoiceInput()) + expect(result.current.status).toBe("idle") + expect(result.current.isSupported).toBe(true) + expect(result.current.interimText).toBe("") + }) + + it("starts listening on startListening() in web-speech mode", async () => { + const onTranscript = vi.fn() + const { result } = renderHook(() => useVoiceInput({ onTranscript })) + + await act(async () => { + await result.current.startListening() + }) + + expect(mockRecognitionInstance.start).toHaveBeenCalled() + expect(result.current.status).toBe("listening") + expect(result.current.mode).toBe("web-speech") + }) + + it("delivers finalized speech to onTranscript", async () => { + const onTranscript = vi.fn() + const { result } = renderHook(() => useVoiceInput({ onTranscript })) + + await act(async () => { + await result.current.startListening() + }) + + // Simulate recognition onresult event + act(() => { + mockRecognitionInstance.onresult({ + resultIndex: 0, + results: [ + Object.assign([[{ transcript: "你好,知夏" }]], { + isFinal: true, + 0: { transcript: "你好,知夏" }, + }), + ], + }) + }) + + expect(onTranscript).toHaveBeenCalledWith("你好,知夏", true) + }) + + it("handles interim results and flushes them on stopListening()", async () => { + const onTranscript = vi.fn() + const onInterim = vi.fn() + const { result } = renderHook(() => + useVoiceInput({ onTranscript, onInterimTranscript: onInterim }) + ) + + await act(async () => { + await result.current.startListening() + }) + + act(() => { + mockRecognitionInstance.onresult({ + resultIndex: 0, + results: [ + Object.assign([[{ transcript: "正在说话" }]], { + isFinal: false, + 0: { transcript: "正在说话" }, + }), + ], + }) + }) + + expect(result.current.interimText).toBe("正在说话") + expect(onInterim).toHaveBeenCalledWith("正在说话") + + // Now stop listening: remaining interim text is flushed + act(() => { + result.current.stopListening() + }) + + expect(onTranscript).toHaveBeenCalledWith("正在说话", true) + expect(result.current.status).toBe("idle") + }) + + it("handles permission denial error gracefully", async () => { + const onError = vi.fn() + const { result } = renderHook(() => useVoiceInput({ onError })) + + await act(async () => { + await result.current.startListening() + }) + + act(() => { + mockRecognitionInstance.onerror({ error: "not-allowed" }) + }) + + expect(result.current.status).toBe("error") + expect(result.current.errorMessage).toBe("micPermissionDenied") + expect(onError).toHaveBeenCalledWith("micPermissionDenied") + }) + + it("cancels listening and aborts recognition without output", async () => { + const onTranscript = vi.fn() + const { result } = renderHook(() => useVoiceInput({ onTranscript })) + + await act(async () => { + await result.current.startListening() + }) + + act(() => { + result.current.cancel() + }) + + expect(mockRecognitionInstance.abort).toHaveBeenCalled() + expect(result.current.status).toBe("idle") + expect(onTranscript).not.toHaveBeenCalled() + }) +}) diff --git a/src/components/chat/composer/use-voice-input.ts b/src/components/chat/composer/use-voice-input.ts new file mode 100644 index 0000000000..9d97d2b3b9 --- /dev/null +++ b/src/components/chat/composer/use-voice-input.ts @@ -0,0 +1,489 @@ +import { useCallback, useEffect, useRef, useState } from "react" + +export type VoiceInputStatus = "idle" | "listening" | "transcribing" | "error" +export type VoiceInputMode = "web-speech" | "gemini" | "whisper" +export type VoiceInputEngine = "auto" | "gemini" | "web-speech" | "whisper" + +export function encodeWav(samples: Float32Array, sampleRate: number): Blob { + const buffer = new ArrayBuffer(44 + samples.length * 2) + const view = new DataView(buffer) + + const writeString = (offset: number, str: string) => { + for (let i = 0; i < str.length; i++) { + view.setUint8(offset + i, str.charCodeAt(i)) + } + } + + writeString(0, "RIFF") + view.setUint32(4, 36 + samples.length * 2, true) + writeString(8, "WAVE") + writeString(12, "fmt ") + view.setUint32(16, 16, true) + view.setUint16(20, 1, true) + view.setUint16(22, 1, true) + view.setUint32(24, sampleRate, true) + view.setUint32(28, sampleRate * 2, true) + view.setUint16(32, 2, true) + view.setUint16(34, 16, true) + writeString(36, "data") + view.setUint32(40, samples.length * 2, true) + + let offset = 44 + for (let i = 0; i < samples.length; i++, offset += 2) { + const s = Math.max(-1, Math.min(1, samples[i])) + view.setInt16(offset, s < 0 ? s * 0x8000 : s * 0x7fff, true) + } + + return new Blob([view], { type: "audio/wav" }) +} + +export async function transcribeAudioWithGemini( + wavBlob: Blob, + endpoint = "http://127.0.0.1:8318/v1/chat/completions", + apiKey = "YaoI3_nkcqDk4otG0S8b5xnpz9kJg6yVL3sjj-e6Tqg" +): Promise { + const base64Data = await new Promise((resolve, reject) => { + const reader = new FileReader() + reader.onloadend = () => { + const result = reader.result as string + const base64 = result.split(",")[1] + resolve(base64) + } + reader.onerror = reject + reader.readAsDataURL(wavBlob) + }) + + const payload = { + model: "gemini-3.8-flash-high", + messages: [ + { + role: "user", + content: [ + { + type: "text", + text: + "你是一个高精度专业语音识别(ASR)引擎。请将这段音频逐字转写为纯文本。\n严格规则:\n" + + "1. 仅输出转写后的文本内容,严禁包含任何解释、问候、代码块或前后缀;\n" + + "2. 严格忠于原文发音。中文输出为规范汉字,英文术语与代码名词(如 Python, Git, PR, Bug, React, API, Token, Hook, TypeScript 等)如实保留标准英文与大小写;\n" + + "3. 适当添加规范标点符号;\n" + + "4. 若音频静音或无清晰人声,输出空字符串。" + }, + { + type: "input_audio", + input_audio: { + data: base64Data, + format: "wav" + } + } + ] + } + ], + temperature: 0.1 + } + + const res = await fetch(endpoint, { + method: "POST", + headers: { + "Content-Type": "application/json", + Authorization: "Bearer " + apiKey + }, + body: JSON.stringify(payload) + }) + + if (!res.ok) { + throw new Error("Gemini ASR HTTP " + res.status) + } + + const data = await res.json() + return data.choices?.[0]?.message?.content?.trim() || "" +} + +export interface UseVoiceInputOptions { + /** Called when a speech segment is transcribed. */ + onTranscript?: (text: string, isFinal: boolean) => void + /** Called with the real-time interim recognition text. */ + onInterimTranscript?: (text: string) => void + /** Called when an error occurs during speech recognition. */ + onError?: (error: string) => void + /** BCP 47 language tag (e.g. 'zh-CN', 'en-US'). Defaults to navigator.language. */ + lang?: string + /** Voice recognition engine choice: 'gemini' (local 8318), 'web-speech', or 'whisper'. */ + engine?: VoiceInputEngine + /** Optional Whisper/ASR transcription API endpoint (e.g. '/v1/audio/transcriptions'). */ + asrEndpoint?: string + /** Optional API token for the ASR endpoint. */ + asrApiKey?: string +} + +export interface UseVoiceInputReturn { + status: VoiceInputStatus + mode: VoiceInputMode + isSupported: boolean + volume: number + interimText: string + errorMessage: string | null + startListening: () => Promise + stopListening: () => void + cancel: () => void +} + +export function useVoiceInput(options: UseVoiceInputOptions = {}): UseVoiceInputReturn { + const { onTranscript, onInterimTranscript, onError, lang, engine = "auto", asrEndpoint, asrApiKey } = options + + const [status, setStatus] = useState("idle") + const [mode, setMode] = useState("web-speech") + const [volume, setVolume] = useState(0) + const [interimText, setInterimText] = useState("") + const [errorMessage, setErrorMessage] = useState(null) + + const recognitionRef = useRef(null) + const mediaStreamRef = useRef(null) + const mediaRecorderRef = useRef(null) + const audioChunksRef = useRef([]) + const pcmChunksRef = useRef([]) + const processorRef = useRef(null) + const sampleRateRef = useRef(16000) + const audioContextRef = useRef(null) + const animationFrameRef = useRef(null) + const isMountedRef = useRef(true) + + useEffect(() => { + isMountedRef.current = true + return () => { + isMountedRef.current = false + } + }, []) + + // Check if Web Speech API or MediaDevices is supported + const isSupported = typeof window !== "undefined" && Boolean( + ("SpeechRecognition" in window || "webkitSpeechRecognition" in window) || + (navigator.mediaDevices && typeof navigator.mediaDevices.getUserMedia === "function") + ) + + const cleanupAudio = useCallback(() => { + if (animationFrameRef.current !== null) { + cancelAnimationFrame(animationFrameRef.current) + animationFrameRef.current = null + } + if (audioContextRef.current && audioContextRef.current.state !== "closed") { + void audioContextRef.current.close().catch(() => {}) + audioContextRef.current = null + } + if (mediaStreamRef.current) { + mediaStreamRef.current.getTracks().forEach((track) => track.stop()) + mediaStreamRef.current = null + } + setVolume(0) + }, []) + + const cancel = useCallback(() => { + if (processorRef.current) { + try { + processorRef.current.disconnect() + } catch {} + processorRef.current = null + } + pcmChunksRef.current = [] + if (recognitionRef.current) { + try { + recognitionRef.current.abort() + } catch {} + recognitionRef.current = null + } + if (mediaRecorderRef.current && mediaRecorderRef.current.state !== "inactive") { + try { + mediaRecorderRef.current.stop() + } catch {} + mediaRecorderRef.current = null + } + cleanupAudio() + setInterimText("") + setStatus("idle") + }, [cleanupAudio]) + + const startVolumeMeter = useCallback((stream: MediaStream) => { + try { + const AudioContextClass = window.AudioContext || (window as any).webkitAudioContext + if (!AudioContextClass) return + const audioCtx = new AudioContextClass() + audioContextRef.current = audioCtx + const source = audioCtx.createMediaStreamSource(stream) + const analyser = audioCtx.createAnalyser() + analyser.fftSize = 256 + source.connect(analyser) + + const dataArray = new Uint8Array(analyser.frequencyBinCount) + const checkVolume = () => { + if (!isMountedRef.current) return + analyser.getByteFrequencyData(dataArray) + let sum = 0 + for (let i = 0; i < dataArray.length; i++) { + sum += dataArray[i] + } + const avg = sum / dataArray.length + const norm = Math.min(1, Math.max(0, avg / 80)) + setVolume(norm) + animationFrameRef.current = requestAnimationFrame(checkVolume) + } + checkVolume() + } catch (e) { + console.warn("[useVoiceInput] Volume meter failed:", e) + } + }, []) + + const startListening = useCallback(async () => { + setErrorMessage(null) + setInterimText("") + + if (engine === "gemini" && typeof window !== "undefined" && navigator.mediaDevices?.getUserMedia) { + setMode("gemini") + try { + const stream = await navigator.mediaDevices.getUserMedia({ audio: true }) + mediaStreamRef.current = stream + startVolumeMeter(stream) + + const AudioContextClass = window.AudioContext || (window as any).webkitAudioContext + const audioCtx = new AudioContextClass() + audioContextRef.current = audioCtx + sampleRateRef.current = audioCtx.sampleRate || 16000 + + const source = audioCtx.createMediaStreamSource(stream) + const processor = audioCtx.createScriptProcessor(4096, 1, 1) + processorRef.current = processor + pcmChunksRef.current = [] + + processor.onaudioprocess = (e: any) => { + if (!isMountedRef.current) return + const input = e.inputBuffer.getChannelData(0) + pcmChunksRef.current.push(new Float32Array(input)) + } + + source.connect(processor) + processor.connect(audioCtx.destination) + + setStatus("listening") + setInterimText("正在聆听中英文... 说完点击停止") + } catch (err: any) { + console.error("[useVoiceInput] Gemini ASR start failed:", err) + const msg = err.name === "NotAllowedError" ? "micPermissionDenied" : err.message + setErrorMessage(msg) + onError?.(msg) + setStatus("error") + cleanupAudio() + } + return + } + + const SpeechRecognitionClass = (typeof window !== "undefined" && ((window as any).SpeechRecognition || (window as any).webkitSpeechRecognition)) || null + + if (SpeechRecognitionClass && !asrEndpoint) { + setMode("web-speech") + try { + if (navigator.mediaDevices?.getUserMedia) { + try { + const stream = await navigator.mediaDevices.getUserMedia({ audio: true }) + mediaStreamRef.current = stream + startVolumeMeter(stream) + } catch {} + } + + const recognition = new SpeechRecognitionClass() + recognitionRef.current = recognition + recognition.continuous = true + recognition.interimResults = true + recognition.maxAlternatives = 1 + // Priority: explicit lang > app Chinese default > navigator.language + recognition.lang = lang || "zh-CN" + + recognition.onresult = (event: any) => { + let finalChunk = "" + let currentInterim = "" + + for (let i = event.resultIndex; i < event.results.length; i++) { + const transcript = event.results[i][0].transcript + if (event.results[i].isFinal) { + finalChunk += transcript + } else { + currentInterim += transcript + } + } + + if (finalChunk) { + const cleaned = finalChunk.replace(/([\u4e00-\u9fa5])\s+([\u4e00-\u9fa5])/g, "$1$2") + onTranscript?.(cleaned, true) + } + const cleanedInterim = currentInterim.replace(/([\u4e00-\u9fa5])\s+([\u4e00-\u9fa5])/g, "$1$2") + setInterimText(cleanedInterim) + onInterimTranscript?.(cleanedInterim) + } + + recognition.onerror = (event: any) => { + console.warn("[useVoiceInput] recognition error:", event.error) + if (event.error === "no-speech") return + const err = event.error === "not-allowed" ? "micPermissionDenied" : event.error + setErrorMessage(err) + onError?.(err) + setStatus("error") + cleanupAudio() + } + + recognition.onend = () => { + cleanupAudio() + setStatus("idle") + setInterimText("") + } + + recognition.start() + setStatus("listening") + } catch (err: any) { + console.error("[useVoiceInput] start failed:", err) + setErrorMessage(err?.message || "Failed to start speech recognition") + setStatus("error") + cleanupAudio() + } + return + } + + if (navigator.mediaDevices?.getUserMedia) { + setMode("whisper") + try { + const stream = await navigator.mediaDevices.getUserMedia({ audio: true }) + mediaStreamRef.current = stream + startVolumeMeter(stream) + + const recorder = new MediaRecorder(stream) + mediaRecorderRef.current = recorder + audioChunksRef.current = [] + + recorder.ondataavailable = (e) => { + if (e.data && e.data.size > 0) { + audioChunksRef.current.push(e.data) + } + } + + recorder.start(250) + setStatus("listening") + } catch (err: any) { + const msg = err.name === "NotAllowedError" ? "micPermissionDenied" : err.message + setErrorMessage(msg) + onError?.(msg) + setStatus("error") + cleanupAudio() + } + return + } + + setErrorMessage("speechNotSupported") + onError?.("speechNotSupported") + setStatus("error") + }, [asrEndpoint, lang, onTranscript, onInterimTranscript, onError, startVolumeMeter, cleanupAudio]) + + const stopListening = useCallback(() => { + if (mode === "gemini") { + setStatus("transcribing") + setInterimText("⚡ 正在通过 Gemini 识别中英文...") + + if (processorRef.current) { + try { + processorRef.current.disconnect() + } catch {} + processorRef.current = null + } + cleanupAudio() + + void (async () => { + try { + const chunks = pcmChunksRef.current + const totalSamples = chunks.reduce((acc, c) => acc + c.length, 0) + if (totalSamples > 0) { + const merged = new Float32Array(totalSamples) + let offset = 0 + for (const c of chunks) { + merged.set(c, offset) + offset += c.length + } + pcmChunksRef.current = [] + + const wavBlob = encodeWav(merged, sampleRateRef.current || 16000) + const transcript = await transcribeAudioWithGemini(wavBlob) + if (transcript) { + onTranscript?.(transcript, true) + } + } + } catch (err: any) { + console.warn("[useVoiceInput] Gemini transcription error:", err) + onError?.(err?.message || "Gemini ASR failed") + } finally { + setStatus("idle") + setInterimText("") + } + })() + return + } + + if (recognitionRef.current) { + try { + recognitionRef.current.stop() + } catch {} + if (interimText.trim()) { + onTranscript?.(interimText.trim(), true) + setInterimText("") + } + cleanupAudio() + setStatus("idle") + return + } + + if (mediaRecorderRef.current && mediaRecorderRef.current.state !== "inactive") { + setStatus("transcribing") + mediaRecorderRef.current.onstop = async () => { + cleanupAudio() + const audioBlob = new Blob(audioChunksRef.current, { type: "audio/webm" }) + audioChunksRef.current = [] + + if (asrEndpoint) { + try { + const formData = new FormData() + formData.append("file", audioBlob, "audio.webm") + formData.append("model", "whisper-1") + formData.append("language", lang?.startsWith("zh") ? "zh" : "en") + + const res = await fetch(asrEndpoint, { + method: "POST", + headers: asrApiKey ? { Authorization: "Bearer " + asrApiKey } : {}, + body: formData, + }) + if (!res.ok) throw new Error("ASR HTTP " + res.status) + const data = await res.json() + const text = data.text || "" + if (text) { + onTranscript?.(text, true) + } + } catch (e: any) { + setErrorMessage(e.message) + onError?.(e.message) + } + } + setStatus("idle") + } + try { + mediaRecorderRef.current.stop() + } catch { + cleanupAudio() + setStatus("idle") + } + } + }, [interimText, onTranscript, asrEndpoint, asrApiKey, lang, onError, cleanupAudio]) + + return { + status, + mode, + isSupported, + volume, + interimText, + errorMessage, + startListening, + stopListening, + cancel, + } +} diff --git a/src/components/chat/message-input.tsx b/src/components/chat/message-input.tsx index 7679ab868f..a93f59c03e 100644 --- a/src/components/chat/message-input.tsx +++ b/src/components/chat/message-input.tsx @@ -1,7 +1,7 @@ "use client" import { useCallback, useEffect, useMemo, useRef, useState } from "react" -import { useTranslations } from "next-intl" +import { useTranslations, useLocale } from "next-intl" import { isImeCompositionKey } from "@/lib/ime-composition" import { Button } from "@/components/ui/button" import { @@ -141,6 +141,8 @@ import { ComposerAddMenu } from "@/components/chat/composer/composer-add-menu" import { ComposerImageThumbnails } from "@/components/chat/composer/composer-image-thumbnails" import { useComposerAttachments } from "@/components/chat/composer/use-composer-attachments" import { useComposerShortcuts } from "@/components/chat/composer/use-composer-shortcuts" +import { useVoiceInput } from "@/components/chat/composer/use-voice-input" +import { ComposerVoiceButton } from "@/components/chat/composer/composer-voice-button" /** * Payload pushed into the composer from outside (e.g. a welcome-page quick @@ -347,6 +349,7 @@ export function MessageInput({ }: MessageInputProps) { const t = useTranslations("Folder.chat.messageInput") const tQueue = useTranslations("Folder.chat.messageQueue") + const locale = useLocale() // Kept as a separate binding from `t` so its call sites — exclusively // upload / attachment toasts — read as a single coherent group when // scanning the file. Same namespace, no extra runtime cost. @@ -379,6 +382,20 @@ export function MessageInput({ setComposerEmpty(ed ? isComposerEmpty(ed) : true) }, []) + const handleVoiceTranscript = useCallback((text: string) => { + if (!text) return + const editor = editorRef.current?.getEditor() + if (!editor) return + editor.chain().focus().insertContent(text).run() + setComposerEmpty(isComposerEmpty(editor)) + }, []) + + const voice = useVoiceInput({ + engine: "gemini", + lang: locale?.toLowerCase().startsWith("zh") ? "zh-CN" : (locale || "zh-CN"), + onTranscript: handleVoiceTranscript, + }) + // Attachments (images → thumbnail strip, files → inline badges) and the "+" // menu's insertable shortcuts. Both are shared with the to-do task composers, // so paste/drop/pick and the skill/quick-message entries behave identically @@ -2040,7 +2057,17 @@ export function MessageInput({ )} -
{actionButtons}
+
+ + {actionButtons} +
{showDragActive && (
diff --git a/src/i18n/messages/en.json b/src/i18n/messages/en.json index 8fbad58f36..d036282f17 100644 --- a/src/i18n/messages/en.json +++ b/src/i18n/messages/en.json @@ -2994,6 +2994,11 @@ "noModels": "No models found", "cancel": "Cancel", "send": "Send", + "voiceInput": "Voice input (click to start)", + "voiceListening": "Listening... Click to finish (Esc to cancel)", + "voiceTranscribing": "Transcribing...", + "micPermissionDenied": "Microphone permission denied. Please allow microphone access in settings.", + "speechNotSupported": "Speech recognition is not supported in this environment", "queueMessage": "Queue message", "steerIntoTurn": "Insert into current turn", "steerAsNote": "Send note for next check", diff --git a/src/i18n/messages/zh-CN.json b/src/i18n/messages/zh-CN.json index d777b9aa14..3880198c4a 100644 --- a/src/i18n/messages/zh-CN.json +++ b/src/i18n/messages/zh-CN.json @@ -2994,6 +2994,11 @@ "noModels": "未找到模型", "cancel": "取消", "send": "发送", + "voiceInput": "语音输入 (点击开始)", + "voiceListening": "正在录音... 点击完成 (Esc 取消)", + "voiceTranscribing": "正在转写...", + "micPermissionDenied": "麦克风权限被拒绝,请在系统或浏览器设置中允许访问", + "speechNotSupported": "当前环境暂不支持语音输入", "queueMessage": "加入队列", "steerIntoTurn": "插入当前回合", "steerAsNote": "发送留言,供下次检查时读取",