"use client"; import { useCallback, useEffect, useRef, useState, useSyncExternalStore } from "react"; import { splitSpeechTextIntoChunks, type AgentSpeakOptions, type AgentSpeechState } from "../speech"; interface BrowserSpeechRecognitionEvent extends Event { readonly resultIndex: number; readonly results: SpeechRecognitionResultList; } interface BrowserSpeechRecognitionErrorEvent extends Event { readonly error: string; readonly message?: string; } interface BrowserSpeechRecognition extends EventTarget { lang: string; continuous: boolean; interimResults: boolean; onresult: ((event: BrowserSpeechRecognitionEvent) => void) | null; onerror: ((event: BrowserSpeechRecognitionErrorEvent) => void) | null; onend: (() => void) | null; start(): void; stop(): void; abort(): void; } type BrowserSpeechRecognitionConstructor = new () => BrowserSpeechRecognition; declare global { interface Window { SpeechRecognition?: BrowserSpeechRecognitionConstructor; webkitSpeechRecognition?: BrowserSpeechRecognitionConstructor; } } const subscribeToBrowserCapabilities = () => () => undefined; function getSpeechSynthesisSupport() { return ( typeof window.Audio !== "undefined" && typeof window.URL !== "undefined" && typeof window.fetch !== "undefined" ); } function getSpeechRecognitionSupport() { return "SpeechRecognition" in window || "webkitSpeechRecognition" in window; } function getServerBrowserCapability() { return false; } export function useAgentSpeechSynthesis() { const [speechState, setSpeechState] = useState("idle"); const [speakingMessageId, setSpeakingMessageId] = useState(null); const audioRef = useRef(null); const currentAudioUrlRef = useRef(null); const audioUrlsRef = useRef>(new Set()); const fetchControllersRef = useRef>(new Set()); const chunkUrlCacheRef = useRef>(new Map()); const chunkPromisesRef = useRef>>(new Map()); const chunksRef = useRef([]); const currentChunkIndexRef = useRef(0); const playbackTokenRef = useRef(0); const activeMessageIdRef = useRef(null); const playChunkRef = useRef<(chunkIndex: number, playbackToken: number) => Promise>(async () => undefined); const isSupported = useSyncExternalStore( subscribeToBrowserCapabilities, getSpeechSynthesisSupport, getServerBrowserCapability ); const detachAudio = useCallback((revokeUrl: boolean) => { const audio = audioRef.current; audioRef.current = null; if (audio) { audio.pause(); audio.onended = null; audio.onerror = null; audio.removeAttribute("src"); audio.load(); } const currentUrl = currentAudioUrlRef.current; currentAudioUrlRef.current = null; if (revokeUrl && currentUrl) { URL.revokeObjectURL(currentUrl); audioUrlsRef.current.delete(currentUrl); chunkUrlCacheRef.current.delete(currentChunkIndexRef.current); } }, []); const releaseAudio = useCallback(() => { fetchControllersRef.current.forEach((controller) => controller.abort()); fetchControllersRef.current.clear(); chunkPromisesRef.current.clear(); detachAudio(false); audioUrlsRef.current.forEach((url) => URL.revokeObjectURL(url)); audioUrlsRef.current.clear(); chunkUrlCacheRef.current.clear(); chunksRef.current = []; currentChunkIndexRef.current = 0; activeMessageIdRef.current = null; }, [detachAudio]); const fetchChunkAudio = useCallback((chunkIndex: number, playbackToken: number) => { const cachedUrl = chunkUrlCacheRef.current.get(chunkIndex); if (cachedUrl) return Promise.resolve(cachedUrl); const currentPromise = chunkPromisesRef.current.get(chunkIndex); if (currentPromise) return currentPromise; const text = chunksRef.current[chunkIndex]; if (!text) return Promise.reject(new Error("语音片段不存在")); const controller = new AbortController(); fetchControllersRef.current.add(controller); const promise = fetch("/api/tts/edge", { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify({ text }), signal: controller.signal }) .then(async (response) => { if (!response.ok) { const payload = await response.json().catch(() => null) as { error?: string } | null; throw new Error(payload?.error ?? `语音生成失败 (${response.status})`); } const audioBlob = await response.blob(); if (!audioBlob.size) throw new Error("语音服务返回了空音频"); if (playbackToken !== playbackTokenRef.current) { throw new DOMException("语音播放已取消", "AbortError"); } const objectUrl = URL.createObjectURL(audioBlob); audioUrlsRef.current.add(objectUrl); chunkUrlCacheRef.current.set(chunkIndex, objectUrl); return objectUrl; }) .finally(() => { fetchControllersRef.current.delete(controller); chunkPromisesRef.current.delete(chunkIndex); }); chunkPromisesRef.current.set(chunkIndex, promise); return promise; }, []); const prefetchChunk = useCallback((chunkIndex: number, playbackToken: number) => { if (chunkIndex >= chunksRef.current.length) return; void fetchChunkAudio(chunkIndex, playbackToken).catch(() => undefined); }, [fetchChunkAudio]); const playChunk = useCallback(async (chunkIndex: number, playbackToken: number) => { setSpeechState("loading"); const objectUrl = await fetchChunkAudio(chunkIndex, playbackToken); if (playbackToken !== playbackTokenRef.current) return; detachAudio(true); currentChunkIndexRef.current = chunkIndex; const audio = new Audio(objectUrl); audio.preload = "auto"; audioRef.current = audio; currentAudioUrlRef.current = objectUrl; audio.onended = () => { if (playbackToken !== playbackTokenRef.current) return; detachAudio(true); const nextChunkIndex = chunkIndex + 1; if (nextChunkIndex >= chunksRef.current.length) { releaseAudio(); setSpeechState("idle"); setSpeakingMessageId(null); return; } currentChunkIndexRef.current = nextChunkIndex; void playChunkRef.current(nextChunkIndex, playbackToken); }; audio.onerror = () => { if (playbackToken !== playbackTokenRef.current) return; playbackTokenRef.current += 1; releaseAudio(); setSpeechState("idle"); setSpeakingMessageId(null); }; await audio.play(); if (playbackToken !== playbackTokenRef.current) return; setSpeechState("playing"); prefetchChunk(chunkIndex + 1, playbackToken); }, [detachAudio, fetchChunkAudio, prefetchChunk, releaseAudio]); useEffect(() => { playChunkRef.current = playChunk; }, [playChunk]); const speak = useCallback(async (messageId: string, text: string, options: AgentSpeakOptions = {}) => { const normalizedText = text.trim(); if (!isSupported || !normalizedText) return; const startOffset = Math.max(0, Math.min(options.startOffset ?? 0, normalizedText.length)); const chunks = splitSpeechTextIntoChunks(normalizedText.slice(startOffset)); if (!chunks.length) return; const playbackToken = playbackTokenRef.current + 1; playbackTokenRef.current = playbackToken; releaseAudio(); chunksRef.current = chunks; activeMessageIdRef.current = messageId; setSpeakingMessageId(messageId); setSpeechState("loading"); try { await playChunk(0, playbackToken); } catch (error) { if (error instanceof DOMException && error.name === "AbortError") return; releaseAudio(); setSpeechState("idle"); setSpeakingMessageId(null); } }, [isSupported, playChunk, releaseAudio]); const pause = useCallback(() => { if (!audioRef.current || speechState !== "playing") return; audioRef.current.pause(); setSpeechState("paused"); }, [speechState]); const resume = useCallback(() => { const audio = audioRef.current; if (!audio) return; setSpeakingMessageId(activeMessageIdRef.current); setSpeechState("playing"); void audio.play().catch(() => { playbackTokenRef.current += 1; releaseAudio(); setSpeechState("idle"); setSpeakingMessageId(null); }); }, [releaseAudio]); const stop = useCallback(() => { playbackTokenRef.current += 1; releaseAudio(); setSpeechState("idle"); setSpeakingMessageId(null); }, [releaseAudio]); useEffect(() => () => { playbackTokenRef.current += 1; releaseAudio(); }, [releaseAudio]); return { speechState, speakingMessageId, speak, pause, resume, stop, isSupported }; } export function useAgentSpeechRecognition(onResult: (text: string) => void) { const [isListening, setIsListening] = useState(false); const recognitionRef = useRef(null); const onResultRef = useRef(onResult); const onErrorRef = useRef<(message: string) => void>(() => undefined); useEffect(() => { onResultRef.current = onResult; }, [onResult]); const isSupported = useSyncExternalStore( subscribeToBrowserCapabilities, getSpeechRecognitionSupport, getServerBrowserCapability ); const start = useCallback((onError?: (message: string) => void) => { if (!isSupported || recognitionRef.current) return; const Recognition = window.SpeechRecognition ?? window.webkitSpeechRecognition; if (!Recognition) return; onErrorRef.current = onError ?? (() => undefined); const recognition = new Recognition(); recognition.lang = "zh-CN"; recognition.continuous = true; recognition.interimResults = false; recognition.onresult = (event) => { for (let index = event.resultIndex; index < event.results.length; index += 1) { if (event.results[index].isFinal) { onResultRef.current(event.results[index][0].transcript); } } }; recognition.onerror = (event) => { setIsListening(false); if (event.error !== "aborted") { onErrorRef.current(getSpeechRecognitionErrorMessage(event.error)); } }; recognition.onend = () => { if (recognitionRef.current === recognition) { recognitionRef.current = null; } setIsListening(false); }; recognitionRef.current = recognition; try { recognition.start(); setIsListening(true); } catch { recognitionRef.current = null; setIsListening(false); onErrorRef.current("无法启动语音输入,请检查浏览器麦克风权限后重试"); } }, [isSupported]); const stop = useCallback(() => { recognitionRef.current?.stop(); recognitionRef.current = null; setIsListening(false); }, []); useEffect(() => () => recognitionRef.current?.abort(), []); return { isListening, start, stop, isSupported }; } function getSpeechRecognitionErrorMessage(error: string) { switch (error) { case "not-allowed": case "service-not-allowed": return "无法使用麦克风,请在浏览器地址栏允许麦克风权限后重试"; case "audio-capture": return "未检测到可用麦克风,请检查设备连接"; case "network": return "语音识别服务连接失败,请检查网络后重试"; case "no-speech": return "未检测到语音,请靠近麦克风后重试"; default: return "语音输入暂时不可用,请稍后重试"; } }