From b7c911f01d3849a96099e620301c9ae23eda31eb Mon Sep 17 00:00:00 2001 From: matevip Date: Sun, 26 Apr 2026 16:37:55 +0800 Subject: [PATCH] feat(stt): DashScope realtime voice + language-aware routing + TalkMode polish MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - DashScope paraformer-realtime-v2 WebSocket streaming - Language-aware provider routing: Whisper for English, Paraformer for Chinese - PCM WAV recording replaces WebM (provider filename bug + diagnostics) - TalkMode push-to-talk fixes (audio drop, WS connecting race) - Vite dev proxy WebSocket upgrade fix - WebSocket binary buffer 8KB → 8MB (Tomcat default truncated voice clips) - Audio chunk pacing at 100ms (DashScope returned 0 chars otherwise) - Resolved language hint propagation + raw frame logging - V46 seed idempotency fix on UI-toggled STT row - Diagnostic cleanup after debugging session --- .../channel/web/TalkModeWebSocketHandler.java | 5 +- .../java/vip/mate/config/WebSocketConfig.java | 32 ++ .../java/vip/mate/stt/AudioMimeTypes.java | 90 ++++ .../main/java/vip/mate/stt/SttProvider.java | 28 +- .../vip/mate/stt/SttProviderRegistry.java | 62 ++- .../main/java/vip/mate/stt/SttService.java | 58 ++- .../java/vip/mate/stt/WavPcmExtractor.java | 69 +++ .../stt/provider/DashScopeSttProvider.java | 483 ++++++++++++++++-- .../mate/stt/provider/OpenAiSttProvider.java | 38 +- .../src/main/resources/db/data-en.sql | 16 + .../src/main/resources/db/data-mysql-en.sql | 20 + .../src/main/resources/db/data-mysql-zh.sql | 19 + .../src/main/resources/db/data-zh.sql | 16 + .../migration/h2/V46__stt_default_enabled.sql | 31 ++ .../mysql/V46__stt_default_enabled.sql | 24 + mateclaw-ui/src/components/chat/TalkMode.vue | 186 +++++-- mateclaw-ui/src/i18n/locales/en-US.ts | 7 +- mateclaw-ui/src/i18n/locales/zh-CN.ts | 7 +- mateclaw-ui/src/utils/wavEncoder.ts | 277 ++++++++++ mateclaw-ui/src/views/Settings/Stt/index.vue | 4 +- mateclaw-ui/vite.config.ts | 7 + 21 files changed, 1374 insertions(+), 105 deletions(-) create mode 100644 mateclaw-server/src/main/java/vip/mate/stt/AudioMimeTypes.java create mode 100644 mateclaw-server/src/main/java/vip/mate/stt/WavPcmExtractor.java create mode 100644 mateclaw-server/src/main/resources/db/migration/h2/V46__stt_default_enabled.sql create mode 100644 mateclaw-server/src/main/resources/db/migration/mysql/V46__stt_default_enabled.sql create mode 100644 mateclaw-ui/src/utils/wavEncoder.ts diff --git a/mateclaw-server/src/main/java/vip/mate/channel/web/TalkModeWebSocketHandler.java b/mateclaw-server/src/main/java/vip/mate/channel/web/TalkModeWebSocketHandler.java index 6f8cea55..45bd703c 100644 --- a/mateclaw-server/src/main/java/vip/mate/channel/web/TalkModeWebSocketHandler.java +++ b/mateclaw-server/src/main/java/vip/mate/channel/web/TalkModeWebSocketHandler.java @@ -118,7 +118,10 @@ public class TalkModeWebSocketHandler extends AbstractWebSocketHandler { sendJson(session, Map.of("type", "state", "state", "processing")); // 2. STT: 音频转文字 - Map sttResult = sttService.transcribe(audioData, "audio.webm", "audio/webm", null); + // 前端用 WavRecorder(Web Audio API + 手写 PCM WAV 编码)— 见 + // mateclaw-ui/src/utils/wavEncoder.ts. WebM/Opus 被 DashScope + // Paraformer 拒收,WAV 是所有 STT provider 都接受的最大公约数。 + Map sttResult = sttService.transcribe(audioData, "audio.wav", "audio/wav", null); if (!Boolean.TRUE.equals(sttResult.get("success"))) { sendJson(session, Map.of("type", "error", "message", "Speech recognition failed: " + sttResult.get("error"))); sendJson(session, Map.of("type", "state", "state", "idle")); diff --git a/mateclaw-server/src/main/java/vip/mate/config/WebSocketConfig.java b/mateclaw-server/src/main/java/vip/mate/config/WebSocketConfig.java index 3c3abeef..79167804 100644 --- a/mateclaw-server/src/main/java/vip/mate/config/WebSocketConfig.java +++ b/mateclaw-server/src/main/java/vip/mate/config/WebSocketConfig.java @@ -1,10 +1,12 @@ package vip.mate.config; import lombok.RequiredArgsConstructor; +import org.springframework.context.annotation.Bean; import org.springframework.context.annotation.Configuration; import org.springframework.web.socket.config.annotation.EnableWebSocket; import org.springframework.web.socket.config.annotation.WebSocketConfigurer; import org.springframework.web.socket.config.annotation.WebSocketHandlerRegistry; +import org.springframework.web.socket.server.standard.ServletServerContainerFactoryBean; import vip.mate.channel.web.TalkModeWebSocketHandler; /** @@ -19,6 +21,19 @@ import vip.mate.channel.web.TalkModeWebSocketHandler; @RequiredArgsConstructor public class WebSocketConfig implements WebSocketConfigurer { + /** + * Max binary frame the TalkMode WebSocket accepts. Recordings come over + * as 16 kHz / 16-bit / mono PCM WAV — roughly 32 KB per second of audio. + * 8 MB headroom covers ~4 minutes of speech, which is more than any + * realistic single-turn voice message. Default is 8 KB which used to + * crash the connection (CloseStatus 1009 "Message too big") on every + * non-trivial recording. + */ + private static final int MAX_BINARY_BUFFER_BYTES = 8 * 1024 * 1024; + + /** Text frames stay reasonably small (init / state / transcript JSON). */ + private static final int MAX_TEXT_BUFFER_BYTES = 64 * 1024; + private final TalkModeWebSocketHandler talkModeHandler; @Override @@ -26,4 +41,21 @@ public class WebSocketConfig implements WebSocketConfigurer { registry.addHandler(talkModeHandler, "/api/v1/talk/ws") .setAllowedOrigins("*"); } + + /** + * Tomcat-level buffer override. Without this Spring's + * {@link org.springframework.web.socket.handler.AbstractWebSocketHandler} + * still buffers the whole binary message before dispatching to the handler + * — capped at 8 KB by default — and a single voice clip blows past that + * limit on the very first send. Bumping the binary buffer is simpler than + * implementing partial-message handling, given the bounded size of + * speech-to-text clips. See discussion in the V46 STT bug-fix series. + */ + @Bean + public ServletServerContainerFactoryBean createWebSocketContainer() { + ServletServerContainerFactoryBean container = new ServletServerContainerFactoryBean(); + container.setMaxBinaryMessageBufferSize(MAX_BINARY_BUFFER_BYTES); + container.setMaxTextMessageBufferSize(MAX_TEXT_BUFFER_BYTES); + return container; + } } diff --git a/mateclaw-server/src/main/java/vip/mate/stt/AudioMimeTypes.java b/mateclaw-server/src/main/java/vip/mate/stt/AudioMimeTypes.java new file mode 100644 index 00000000..98ecadfc --- /dev/null +++ b/mateclaw-server/src/main/java/vip/mate/stt/AudioMimeTypes.java @@ -0,0 +1,90 @@ +package vip.mate.stt; + +import java.util.Locale; +import java.util.Map; + +/** + * Filename inference helper for the OpenAI Whisper STT path. + * + *

Whisper's {@code /v1/audio/transcriptions} endpoint reads the + * filename extension on the multipart {@code file} part to infer + * audio format; the {@code Content-Type} alone isn't enough because + * Hutool's 3-arg {@code form(name, bytes, fileName)} overload deduces + * the multipart Content-Type from the extension we pass. Hence this + * helper picks an extension that matches the actual bytes. + * + *

Previous bug (pre-fix): both providers hardcoded {@code "audio.ogg"} + * as the default filename even when the upstream content was WebM/Opus, + * which DashScope's HTTP path then tried to decode as Ogg and 400'd. + * That bug + DashScope's HTTP STT are both gone now (DashScope went to + * WebSocket); this class survives because Whisper still cares. + */ +public final class AudioMimeTypes { + + /** Fallback when neither contentType nor filename gives us a hint. */ + private static final String DEFAULT_EXTENSION = "wav"; + + /** content-type → conventional file extension. Ordered for documentation only. */ + private static final Map CONTENT_TYPE_TO_EXTENSION = Map.ofEntries( + Map.entry("audio/wav", "wav"), + Map.entry("audio/wave", "wav"), + Map.entry("audio/x-wav", "wav"), + Map.entry("audio/mpeg", "mp3"), + Map.entry("audio/mp3", "mp3"), + Map.entry("audio/mp4", "m4a"), + Map.entry("audio/m4a", "m4a"), + Map.entry("audio/x-m4a", "m4a"), + Map.entry("audio/aac", "aac"), + Map.entry("audio/flac", "flac"), + Map.entry("audio/ogg", "ogg"), + Map.entry("audio/webm", "webm"), + Map.entry("audio/amr", "amr")); + + /** file extension → conventional content-type (the inverse, for filename-first cases). */ + private static final Map EXTENSION_TO_CONTENT_TYPE = Map.ofEntries( + Map.entry("wav", "audio/wav"), + Map.entry("mp3", "audio/mpeg"), + Map.entry("m4a", "audio/mp4"), + Map.entry("mp4", "audio/mp4"), + Map.entry("aac", "audio/aac"), + Map.entry("flac", "audio/flac"), + Map.entry("ogg", "audio/ogg"), + Map.entry("webm", "audio/webm"), + Map.entry("amr", "audio/amr")); + + private AudioMimeTypes() {} + + /** + * Choose a filename for the upload. Order of precedence: + *

    + *
  1. The caller-supplied filename, when it has a known audio extension.
  2. + *
  3. A name synthesised from the content-type, e.g. {@code audio/mpeg → audio.mp3}.
  4. + *
  5. {@code audio.wav} as a final fallback (WAV is universally accepted).
  6. + *
+ */ + public static String resolveFileName(String fileName, String contentType) { + if (fileName != null && !fileName.isBlank() && extensionOf(fileName) != null) { + return fileName; + } + String extension = extensionForContentType(contentType); + return "audio." + (extension != null ? extension : DEFAULT_EXTENSION); + } + + /** Extract the lower-cased extension (without the dot), or null. Package-private for tests. */ + static String extensionOf(String fileName) { + if (fileName == null) return null; + int dot = fileName.lastIndexOf('.'); + if (dot < 0 || dot == fileName.length() - 1) return null; + String ext = fileName.substring(dot + 1).toLowerCase(Locale.ROOT); + return EXTENSION_TO_CONTENT_TYPE.containsKey(ext) ? ext : null; + } + + /** Map a content-type to its conventional extension, or null when unknown. */ + static String extensionForContentType(String contentType) { + if (contentType == null) return null; + // Strip parameters: "audio/webm; codecs=opus" → "audio/webm" + int semi = contentType.indexOf(';'); + String base = (semi >= 0 ? contentType.substring(0, semi) : contentType).trim().toLowerCase(Locale.ROOT); + return CONTENT_TYPE_TO_EXTENSION.get(base); + } +} diff --git a/mateclaw-server/src/main/java/vip/mate/stt/SttProvider.java b/mateclaw-server/src/main/java/vip/mate/stt/SttProvider.java index 63c2e2a4..92c78714 100644 --- a/mateclaw-server/src/main/java/vip/mate/stt/SttProvider.java +++ b/mateclaw-server/src/main/java/vip/mate/stt/SttProvider.java @@ -3,7 +3,16 @@ package vip.mate.stt; import vip.mate.system.model.SystemSettingsDTO; /** - * STT 语音识别提供商接口 + * STT 语音识别提供商接口. + * + *

Auto-detect ordering uses ascending priority (low number = preferred). + * Most providers can use the default {@link #autoDetectOrder()} value, but + * providers with strong language bias should override + * {@link #autoDetectOrder(String)} so the registry picks the right primary + * for the user's locale: OpenAI Whisper is the canonical English path, + * DashScope Paraformer is the canonical Chinese path. Mixing them up at + * dispatch time costs accuracy AND latency (the wrong primary tends to + * produce garbage that the fallback can't easily compensate for). */ public interface SttProvider { String id(); @@ -12,6 +21,23 @@ public interface SttProvider { int autoDetectOrder(); boolean isAvailable(SystemSettingsDTO config); + /** + * Per-language priority hook. Default implementation returns the + * language-agnostic {@link #autoDetectOrder()} value, so existing + * providers stay backwards-compatible. Override when the provider has + * a known language strength — e.g. OpenAI Whisper returns a smaller + * number for {@code "en"} than for {@code "zh"} to win the auto-pick + * for English users. + * + * @param language IETF / ISO-639 language hint, possibly {@code null}. + * Implementations should match conservatively (prefix + * match on {@code zh}, {@code en}, etc.) and gracefully + * fall back to {@link #autoDetectOrder()} on unknown. + */ + default int autoDetectOrder(String language) { + return autoDetectOrder(); + } + /** * 转写音频 */ diff --git a/mateclaw-server/src/main/java/vip/mate/stt/SttProviderRegistry.java b/mateclaw-server/src/main/java/vip/mate/stt/SttProviderRegistry.java index 63468dca..7846e7f9 100644 --- a/mateclaw-server/src/main/java/vip/mate/stt/SttProviderRegistry.java +++ b/mateclaw-server/src/main/java/vip/mate/stt/SttProviderRegistry.java @@ -11,41 +11,85 @@ import java.util.function.Function; import java.util.stream.Collectors; /** - * STT 提供商注册表 + * STT 提供商注册表. + * + *

Provider selection is two-stage: + *

    + *
  1. If the user pinned a specific provider in settings, that wins.
  2. + *
  3. Otherwise sort all available providers by + * {@link SttProvider#autoDetectOrder(String)} — the per-language + * hook lets Whisper win on English while Paraformer wins on Chinese. + * Falls back to the language-agnostic order when no language hint + * is supplied.
  4. + *
+ * + *

The registry is constructed once at startup and the unsorted provider + * list is held — sorting happens on demand because the order depends on the + * incoming language hint. */ @Slf4j @Component public class SttProviderRegistry { - private final List sortedProviders; + private final List providers; private final Map providerMap; public SttProviderRegistry(List providers) { - this.sortedProviders = providers.stream() - .sorted(Comparator.comparingInt(SttProvider::autoDetectOrder)) - .toList(); + this.providers = List.copyOf(providers); this.providerMap = providers.stream() .collect(Collectors.toMap(SttProvider::id, Function.identity())); - log.info("注册 STT 提供商 {} 个: {}", sortedProviders.size(), - sortedProviders.stream().map(p -> p.id() + "(order=" + p.autoDetectOrder() + ")").toList()); + log.info("注册 STT 提供商 {} 个: {}", providers.size(), + providers.stream() + .map(p -> p.id() + "(default-order=" + p.autoDetectOrder() + ")") + .toList()); } + /** Backwards-compatible no-language overload. Prefer the (config, language) form. */ public SttProvider resolve(SystemSettingsDTO config) { + return resolve(config, null); + } + + /** + * Pick the primary provider given the user's settings + a language hint. + * Returns null when nothing is available — callers should treat that as + * "no API key configured anywhere" and surface the actionable error. + */ + public SttProvider resolve(SystemSettingsDTO config, String language) { String configuredId = config.getSttProvider(); if (configuredId != null && !configuredId.isBlank() && !"auto".equals(configuredId)) { SttProvider p = providerMap.get(configuredId); if (p != null && p.isAvailable(config)) return p; + // Configured-but-unavailable falls through to auto so the user + // still gets a result if any other provider has its key set. } - for (SttProvider p : sortedProviders) { + for (SttProvider p : sortedByLanguage(language)) { if (p.isAvailable(config)) return p; } return null; } + /** Backwards-compatible no-language overload. */ public List fallbackCandidates(SystemSettingsDTO config, String excludeId) { - return sortedProviders.stream() + return fallbackCandidates(config, excludeId, null); + } + + /** + * Available providers other than {@code excludeId}, ordered by their + * priority for the given language. The fallback list always uses the + * language-aware order — if Whisper failed on Chinese, Paraformer is + * the right next-best, not whatever happened to come next in the + * default order. + */ + public List fallbackCandidates(SystemSettingsDTO config, String excludeId, String language) { + return sortedByLanguage(language).stream() .filter(p -> !p.id().equals(excludeId)) .filter(p -> p.isAvailable(config)) .toList(); } + + private List sortedByLanguage(String language) { + return providers.stream() + .sorted(Comparator.comparingInt(p -> p.autoDetectOrder(language))) + .toList(); + } } diff --git a/mateclaw-server/src/main/java/vip/mate/stt/SttService.java b/mateclaw-server/src/main/java/vip/mate/stt/SttService.java index 44432bac..c7da7d8d 100644 --- a/mateclaw-server/src/main/java/vip/mate/stt/SttService.java +++ b/mateclaw-server/src/main/java/vip/mate/stt/SttService.java @@ -26,6 +26,15 @@ public class SttService { return Map.of("success", false, "error", "STT 功能未启用,请在系统设置中开启"); } + // Per-call dispatch trace — without this the only signal that STT + // is even being attempted is the eventual provider success/failure + // log, which makes "no audio reached us" indistinguishable from + // "audio reached us but provider rejected it". + int bytes = audioData != null ? audioData.length : 0; + log.info("[STT] dispatch bytes={} fileName={} contentType={} language={} provider={}", + bytes, fileName, contentType, language, + config.getSttProvider() != null ? config.getSttProvider() : "auto"); + SttRequest request = SttRequest.builder() .audioData(audioData) .fileName(fileName) @@ -45,26 +54,61 @@ public class SttService { } private SttResult transcribeWithFallback(SttRequest request, SystemSettingsDTO config) { - SttProvider primary = providerRegistry.resolve(config); - if (primary == null) { - return SttResult.failure("没有可用的 STT Provider,请检查配置"); + // Language hint resolution order: + // 1. Caller-supplied request.language (explicit per-call override) + // 2. UI language from system settings (zh-CN / en-US) + // 3. null — registry falls back to the language-agnostic order + // Both registry primary-pick and fallback-candidate ordering use the + // same hint; without it Paraformer/Whisper would frequently swap + // priorities mid-fallback for the same conversation. + // + // Critical: write the resolved hint BACK into the request, so + // providers (especially DashScope's run-task language_hints field) + // see it. Pre-fix, providers got null even though we routed by + // zh-CN — DashScope's auto-detect would then sit through 2-3s of + // Chinese audio and emit zero result-generated events. + String languageHint = request.getLanguage(); + if (languageHint == null || languageHint.isBlank()) { + languageHint = config.getLanguage(); + if (languageHint != null && !languageHint.isBlank()) { + request.setLanguage(languageHint); + } } + SttProvider primary = providerRegistry.resolve(config, languageHint); + if (primary == null) { + // Most common cause: no provider has its API key configured. Tell + // the operator that explicitly so they don't dig through provider + // logs looking for the real reason. + log.warn("[STT] no provider available — check DashScope / OpenAI API keys in 模型管理"); + return SttResult.failure("没有可用的 STT Provider,请在模型管理中配置 DashScope 或 OpenAI API Key"); + } + log.info("[STT] primary provider resolved: {} (language={})", primary.id(), languageHint); SttResult result = primary.transcribe(request, config); - if (result.isSuccess()) return result; + if (result.isSuccess()) { + log.info("[STT] success via {} ({} chars)", primary.id(), result.getText() != null ? result.getText().length() : 0); + return result; + } + log.warn("[STT] primary {} failed: {}", primary.id(), result.getErrorMessage()); List errors = new ArrayList<>(); errors.add(primary.id() + ": " + result.getErrorMessage()); if (Boolean.TRUE.equals(config.getSttFallbackEnabled())) { - for (SttProvider fb : providerRegistry.fallbackCandidates(config, primary.id())) { - log.info("[STT] Trying fallback provider: {}", fb.id()); + for (SttProvider fb : providerRegistry.fallbackCandidates(config, primary.id(), languageHint)) { + log.info("[STT] trying fallback provider: {}", fb.id()); result = fb.transcribe(request, config); - if (result.isSuccess()) return result; + if (result.isSuccess()) { + log.info("[STT] fallback success via {} ({} chars)", fb.id(), + result.getText() != null ? result.getText().length() : 0); + return result; + } + log.warn("[STT] fallback {} failed: {}", fb.id(), result.getErrorMessage()); errors.add(fb.id() + ": " + result.getErrorMessage()); } } + log.error("[STT] all providers failed — errors: {}", errors); return SttResult.failure("所有 STT Provider 均失败\n" + String.join("\n", errors)); } } diff --git a/mateclaw-server/src/main/java/vip/mate/stt/WavPcmExtractor.java b/mateclaw-server/src/main/java/vip/mate/stt/WavPcmExtractor.java new file mode 100644 index 00000000..1f634385 --- /dev/null +++ b/mateclaw-server/src/main/java/vip/mate/stt/WavPcmExtractor.java @@ -0,0 +1,69 @@ +package vip.mate.stt; + +import java.nio.ByteBuffer; +import java.nio.ByteOrder; + +/** + * Strip the RIFF/WAVE header off a WAV blob to expose raw PCM samples. + * + *

DashScope's realtime ASR expects the {@code parameters.format = "pcm"} + * input as **bare 16-bit signed little-endian PCM**, not WAV. The frontend + * (see {@code mateclaw-ui/src/utils/wavEncoder.ts}) emits a 16 kHz mono + * 16-bit WAV with the canonical 44-byte header — this helper unwraps it. + * + *

Why not just send the WAV: DashScope rejects with "format mismatch" + * because the first 44 bytes look like garbage when interpreted as PCM + * samples — they're the RIFF magic + format chunk metadata. + * + *

Limitations: handles only the canonical 44-byte WAV layout produced by + * MateClaw's WavRecorder. WAVs with extra chunks (LIST, JUNK, …) before the + * data chunk would need a chunk-walking parser. We don't currently accept + * arbitrary uploads, so the tighter scope is fine; if this changes, + * extend {@link #extract} to scan for the {@code "data"} chunk header + * instead of assuming offset 36. + */ +public final class WavPcmExtractor { + + /** Bytes before the "data" chunk in a canonical mono 16-bit PCM WAV. */ + public static final int CANONICAL_HEADER_BYTES = 44; + + /** Sample rate field offset in the canonical WAV header. */ + private static final int OFFSET_SAMPLE_RATE = 24; + + private WavPcmExtractor() {} + + /** + * Extract raw PCM bytes from a WAV blob. Throws when the input is too short + * or the magic header bytes don't look like RIFF/WAVE — better to fail loud + * here than ship garbage to DashScope and chase a confusing error code. + */ + public static byte[] extract(byte[] wavBytes) { + if (wavBytes == null || wavBytes.length < CANONICAL_HEADER_BYTES) { + throw new IllegalArgumentException( + "WAV input too short: " + (wavBytes == null ? 0 : wavBytes.length) + " bytes"); + } + if (wavBytes[0] != 'R' || wavBytes[1] != 'I' || wavBytes[2] != 'F' || wavBytes[3] != 'F' + || wavBytes[8] != 'W' || wavBytes[9] != 'A' || wavBytes[10] != 'V' || wavBytes[11] != 'E') { + throw new IllegalArgumentException("Not a WAV (missing RIFF/WAVE magic)"); + } + byte[] pcm = new byte[wavBytes.length - CANONICAL_HEADER_BYTES]; + System.arraycopy(wavBytes, CANONICAL_HEADER_BYTES, pcm, 0, pcm.length); + return pcm; + } + + /** + * Read the sample rate from a WAV header. Used by callers that need to + * tell DashScope the actual rate of the audio (the API requires the rate + * up front in the {@code run-task} message — getting it wrong produces + * recognisable but distorted transcripts). + */ + public static int sampleRate(byte[] wavBytes) { + if (wavBytes == null || wavBytes.length < CANONICAL_HEADER_BYTES) { + throw new IllegalArgumentException("WAV input too short for sample-rate read"); + } + return ByteBuffer.wrap(wavBytes, OFFSET_SAMPLE_RATE, 4) + .order(ByteOrder.LITTLE_ENDIAN) + .getInt(); + } + +} diff --git a/mateclaw-server/src/main/java/vip/mate/stt/provider/DashScopeSttProvider.java b/mateclaw-server/src/main/java/vip/mate/stt/provider/DashScopeSttProvider.java index 18e9c5d3..8c4a8415 100644 --- a/mateclaw-server/src/main/java/vip/mate/stt/provider/DashScopeSttProvider.java +++ b/mateclaw-server/src/main/java/vip/mate/stt/provider/DashScopeSttProvider.java @@ -1,7 +1,5 @@ package vip.mate.stt.provider; -import cn.hutool.http.HttpRequest; -import cn.hutool.http.HttpResponse; import com.fasterxml.jackson.databind.JsonNode; import com.fasterxml.jackson.databind.ObjectMapper; import lombok.RequiredArgsConstructor; @@ -11,63 +9,486 @@ import vip.mate.llm.service.ModelProviderService; import vip.mate.stt.SttProvider; import vip.mate.stt.SttRequest; import vip.mate.stt.SttResult; +import vip.mate.stt.WavPcmExtractor; import vip.mate.system.model.SystemSettingsDTO; +import java.net.URI; +import java.net.http.HttpClient; +import java.net.http.WebSocket; +import java.nio.ByteBuffer; +import java.time.Duration; +import java.util.LinkedHashMap; +import java.util.Map; +import java.util.UUID; +import java.util.concurrent.CompletionStage; +import java.util.concurrent.CountDownLatch; +import java.util.concurrent.ExecutionException; +import java.util.concurrent.TimeUnit; +import java.util.concurrent.TimeoutException; +import java.util.concurrent.atomic.AtomicReference; + /** - * DashScope STT Provider — Paraformer(OpenAI 兼容接口) - *

- * 复用模型管理中的 DashScope API Key。中文识别效果优秀。 + * DashScope STT Provider — Paraformer Realtime via WebSocket. + * + *

DashScope's only sync-callable STT path is the realtime WebSocket API + * — there is no /audio/transcriptions endpoint on either the + * native or OpenAI-compatible HTTP surface (verified empirically, returns + * 404). The earlier sync-HTTP version of this provider was speculative and + * has been replaced by this one. + * + *

Wire protocol

+ * Documented at Aliyun DashScope Realtime ASR. Message exchange: + *
    + *
  1. Open WS to {@value #WS_URL} with {@code Authorization: bearer + * } header.
  2. + *
  3. Client sends a {@code run-task} text frame with task_id + + * paraformer-realtime-v2 model + format/sample-rate parameters.
  4. + *
  5. Server replies with {@code task-started} text frame.
  6. + *
  7. Client streams raw 16-bit PCM bytes as binary frames (chunked at + * ~100ms each = {@value #CHUNK_BYTES} bytes for 16 kHz mono).
  8. + *
  9. Server emits {@code result-generated} events as transcripts come + * in. Each event carries a sentence keyed by {@code begin_time}; + * later events with the same {@code begin_time} update the same + * sentence (interim → final).
  10. + *
  11. Client sends {@code finish-task} text frame; server replies with + * {@code task-finished}; both sides close.
  12. + *
+ * + *

The {@link SttProvider} interface is sync — we bridge the async WS + * conversation to a blocking call via {@link CountDownLatch} (run-task ack + * + task-finished ack) plus an overall hard timeout. The whole transcribe + * call returns either a full transcript or a domain-typed + * {@link SttResult#failure} after at most {@value #OVERALL_TIMEOUT_MS}ms. */ @Slf4j @Component @RequiredArgsConstructor public class DashScopeSttProvider implements SttProvider { + /** DashScope WS endpoint for realtime inference (audio/text/multimodal). */ + static final URI WS_URL = URI.create("wss://dashscope.aliyuncs.com/api-ws/v1/inference/"); + + /** Default model — paraformer-realtime-v2 is the canonical 2024+ realtime ASR. */ + static final String DEFAULT_MODEL = "paraformer-realtime-v2"; + + /** Default sample rate in Hz. Must match the actual WAV — the helper reads it. */ + static final int DEFAULT_SAMPLE_RATE_HZ = 16_000; + + /** ~100ms of 16 kHz / 16-bit / mono PCM. DashScope recommends 100-300ms chunks. */ + static final int CHUNK_BYTES = 3200; + + /** + * How long to sleep between chunks. Paraformer-Realtime expects audio to + * arrive at roughly the natural recording rate; if we dump the whole clip + * in tens of milliseconds the server discards the stream and replies with + * task-finished + zero result-generated events. The official Python SDK + * does the same with {@code time.sleep(0.1)} between chunks. Matches + * {@link #CHUNK_BYTES} (100ms of audio → 100ms wall sleep). + */ + static final long CHUNK_PACING_MS = 100L; + + /** How long to wait for the WS handshake + task-started ack before giving up. */ + static final long TASK_STARTED_TIMEOUT_MS = 10_000L; + + /** Overall budget for a single transcribe — beyond this we abort the WS. */ + static final long OVERALL_TIMEOUT_MS = 60_000L; + private final ModelProviderService modelProviderService; private final ObjectMapper objectMapper; - - private static final String BASE_URL = "https://dashscope.aliyuncs.com/compatible-mode/v1"; - private static final String DEFAULT_MODEL = "paraformer-v2"; + /** Shared HttpClient — JDK's WebSocket builder doesn't reuse the underlying + * connection pool when you allocate a fresh client per call, so making + * this a field saves a connection-pool spin-up on every transcribe. */ + private final HttpClient httpClient = HttpClient.newHttpClient(); @Override public String id() { return "dashscope"; } - @Override public String label() { return "DashScope (Paraformer)"; } + @Override public String label() { return "DashScope (Paraformer Realtime)"; } @Override public boolean requiresCredential() { return true; } @Override public int autoDetectOrder() { return 150; } + /** + * Per-language priority. Paraformer is the strongest mainstream Chinese + * STT, so push it ahead of Whisper on zh — see {@link SttProvider} javadoc + * for the routing rationale. + */ + @Override + public int autoDetectOrder(String language) { + if (language == null) return autoDetectOrder(); + String lang = language.toLowerCase(); + if (lang.startsWith("zh")) return 60; + return autoDetectOrder(); + } + @Override public boolean isAvailable(SystemSettingsDTO config) { - try { return modelProviderService.isProviderConfigured("dashscope"); } - catch (Exception e) { return false; } + try { + return modelProviderService.isProviderConfigured("dashscope"); + } catch (Exception e) { + log.warn("[DashScope STT] availability check failed: {}", e.getMessage()); + return false; + } } @Override public SttResult transcribe(SttRequest request, SystemSettingsDTO config) { try { String apiKey = modelProviderService.getProviderConfig("dashscope").getApiKey(); - if (apiKey == null) return SttResult.failure("DashScope API Key 未配置"); - - String model = request.getModel() != null ? request.getModel() : DEFAULT_MODEL; - String fileName = request.getFileName() != null ? request.getFileName() : "audio.ogg"; - - HttpResponse response = HttpRequest.post(BASE_URL + "/audio/transcriptions") - .header("Authorization", "Bearer " + apiKey) - .form("model", model) - .form("file", request.getAudioData(), request.getContentType(), fileName) - .timeout(60_000) - .execute(); - - if (response.getStatus() == 200) { - JsonNode result = objectMapper.readTree(response.body()); - String text = result.path("text").asText(""); - log.info("[DashScope STT] Transcribed {} chars (model={})", text.length(), model); - return SttResult.success(text); - } else { - log.warn("[DashScope STT] Failed: HTTP {} - {}", response.getStatus(), response.body()); - return SttResult.failure("DashScope STT 失败: HTTP " + response.getStatus()); + if (apiKey == null || apiKey.isBlank()) { + return SttResult.failure("DashScope API Key 未配置"); } + byte[] audio = request.getAudioData(); + if (audio == null || audio.length < WavPcmExtractor.CANONICAL_HEADER_BYTES) { + return SttResult.failure("音频为空或过短"); + } + byte[] pcm = WavPcmExtractor.extract(audio); + int sampleRate = WavPcmExtractor.sampleRate(audio); + String model = request.getModel() != null ? request.getModel() : DEFAULT_MODEL; + String taskId = UUID.randomUUID().toString().replace("-", ""); + + // Peak/RMS check — the silence path is a failure mode worth its + // own log line so users can tell "mic captured nothing" from + // "DashScope rejected real audio". Successful calls log peak/rms + // at DEBUG only; a healthy call shouldn't produce a per-request + // INFO log every time the user holds the talk button. + int[] peakRms = computePcmPeakRms(pcm); + if (peakRms[0] == 0) { + log.warn("[DashScope STT] PCM is silent (peak=0, bytes={}) — check mic permission / frontend recording", + pcm.length); + return SttResult.failure( + "音频为静音(PCM peak=0)— 检查麦克风权限或前端录制实现"); + } + log.debug("[DashScope STT] PCM stats — bytes={} samples={} peak={} rms={} sampleRate={}", + pcm.length, pcm.length / 2, peakRms[0], peakRms[1], sampleRate); + + DashScopeSession session = new DashScopeSession(taskId, objectMapper); + WebSocket ws; + try { + ws = httpClient.newWebSocketBuilder() + .header("Authorization", "bearer " + apiKey) + .connectTimeout(Duration.ofMillis(TASK_STARTED_TIMEOUT_MS)) + .buildAsync(WS_URL, session) + .get(TASK_STARTED_TIMEOUT_MS, TimeUnit.MILLISECONDS); + } catch (TimeoutException e) { + return SttResult.failure("DashScope WS 握手超时"); + } + + try { + // 1. run-task. Envelope dumped at DEBUG only — the JSON is + // identical across calls modulo task_id + language hint, so + // logging it on every transcribe just clutters logs. + String runTask = buildRunTask(taskId, model, sampleRate, request.getLanguage()); + log.debug("[DashScope STT] run-task envelope: {}", runTask); + ws.sendText(runTask, true).get(TASK_STARTED_TIMEOUT_MS, TimeUnit.MILLISECONDS); + + // 2. wait for task-started ack + if (!session.awaitTaskStarted(TASK_STARTED_TIMEOUT_MS, TimeUnit.MILLISECONDS)) { + return SttResult.failure("DashScope task-started 超时"); + } + if (session.failed()) { + return SttResult.failure("DashScope: " + session.errorMessage()); + } + + // 3. stream PCM chunks at real-time pace. Paraformer-Realtime + // is built for live mic input and silently drops audio when it + // arrives faster than wall-clock — symptom is 0 chars + // transcribed even though the protocol completes successfully + // (no task-failed). Sleep 100ms between 100ms chunks so total + // send time ≈ audio duration, matching what DashScope's own + // SDK examples do (time.sleep(0.1) per chunk). + int chunksSent = 0; + long sendStart = System.currentTimeMillis(); + for (int offset = 0; offset < pcm.length; offset += CHUNK_BYTES) { + int len = Math.min(CHUNK_BYTES, pcm.length - offset); + ByteBuffer chunk = ByteBuffer.wrap(pcm, offset, len); + ws.sendBinary(chunk, true).get(TASK_STARTED_TIMEOUT_MS, TimeUnit.MILLISECONDS); + chunksSent++; + Thread.sleep(CHUNK_PACING_MS); + // Cheap fail-fast: if the server already said we're done / + // failed mid-stream, stop sending so we don't waste seconds + // sleeping on a dead connection. + if (session.failed() || session.taskFinishedRaised()) break; + } + long sendDuration = System.currentTimeMillis() - sendStart; + log.debug("[DashScope STT] streamed {} chunks ({} bytes) in {} ms", + chunksSent, pcm.length, sendDuration); + + // 4. finish-task + ws.sendText(buildFinishTask(taskId), true).get(TASK_STARTED_TIMEOUT_MS, TimeUnit.MILLISECONDS); + + // 5. wait for task-finished + if (!session.awaitTaskFinished(OVERALL_TIMEOUT_MS, TimeUnit.MILLISECONDS)) { + return SttResult.failure("DashScope task-finished 超时"); + } + if (session.failed()) { + return SttResult.failure("DashScope: " + session.errorMessage()); + } + + String text = session.aggregatedText(); + log.info("[DashScope STT] Transcribed {} chars from {} result-events " + + "(model={}, sampleRate={}, pcmBytes={})", + text.length(), session.resultEventCount(), model, sampleRate, pcm.length); + if (text.isEmpty() && session.resultEventCount() == 0) { + // Distinct failure mode: protocol completed cleanly but + // server never sent a single result-generated event. + // Almost always means the audio was discarded for + // pacing/format reasons. Surface as a typed failure so + // the fallback chain (Whisper) can still try. + return SttResult.failure( + "DashScope 收到 0 个识别事件——可能是音频格式或节奏问题"); + } + return SttResult.success(text); + } finally { + // Best-effort close. abort() is fire-and-forget; we don't need to wait. + try { + ws.sendClose(WebSocket.NORMAL_CLOSURE, "done"); + } catch (Exception ignored) { + ws.abort(); + } + } + } catch (TimeoutException e) { + log.warn("[DashScope STT] timeout: {}", e.getMessage()); + return SttResult.failure("DashScope STT 超时: " + e.getMessage()); + } catch (ExecutionException e) { + Throwable cause = e.getCause() != null ? e.getCause() : e; + log.error("[DashScope STT] WS error: {}", cause.getMessage(), cause); + return SttResult.failure("DashScope STT WS 错误: " + cause.getMessage()); + } catch (InterruptedException e) { + Thread.currentThread().interrupt(); + return SttResult.failure("DashScope STT 被中断"); } catch (Exception e) { log.error("[DashScope STT] Error: {}", e.getMessage(), e); return SttResult.failure("DashScope STT 异常: " + e.getMessage()); } } + + /* ====================================================================== */ + /* Wire-format helpers (package-private for unit testing). */ + /* ====================================================================== */ + + String buildRunTask(String taskId, String model, int sampleRate, String language) throws Exception { + Map parameters = new LinkedHashMap<>(); + parameters.put("format", "pcm"); + parameters.put("sample_rate", sampleRate); + // Language hint when supplied — paraformer-realtime-v2 supports + // "zh", "en", "ja", "ko" via language_hints. Skip when null/blank + // to let the model auto-detect. + if (language != null && !language.isBlank()) { + // Strip locale suffix (zh-CN → zh). + String hint = language.toLowerCase(); + int dash = hint.indexOf('-'); + if (dash > 0) hint = hint.substring(0, dash); + parameters.put("language_hints", new String[]{hint}); + } + + Map payload = Map.of( + "task_group", "audio", + "task", "asr", + "function", "recognition", + "model", model, + "parameters", parameters, + "input", Map.of()); + Map message = Map.of( + "header", Map.of( + "action", "run-task", + "task_id", taskId, + "streaming", "duplex"), + "payload", payload); + return objectMapper.writeValueAsString(message); + } + + /** + * Compute peak (max absolute value) and RMS for 16-bit signed + * little-endian PCM bytes. Returns {peak, rms} as ints for log-friendly + * formatting. Both metrics are in raw int16 units (-32768..32767). + * + *

Reference values for 16-bit PCM at typical recording levels: + *

    + *
  • Silence / muted mic: peak ≤ 5, rms ≤ 2
  • + *
  • Quiet speech: peak ≈ 1000-5000, rms ≈ 200-1000
  • + *
  • Normal speech: peak ≈ 5000-20000, rms ≈ 1000-5000
  • + *
  • Loud / close-mic: peak ≈ 20000-32000, rms ≈ 5000-15000
  • + *
+ */ + static int[] computePcmPeakRms(byte[] pcm) { + if (pcm == null || pcm.length < 2) { + return new int[]{0, 0}; + } + int peak = 0; + long sumSq = 0; + int sampleCount = pcm.length / 2; + for (int i = 0; i < sampleCount; i++) { + // Little-endian 16-bit signed: low byte first. + int lo = pcm[i * 2] & 0xFF; + int hi = pcm[i * 2 + 1]; // signed + int sample = (hi << 8) | lo; + int abs = Math.abs(sample); + if (abs > peak) peak = abs; + sumSq += (long) sample * sample; + } + int rms = (int) Math.sqrt((double) sumSq / sampleCount); + return new int[]{peak, rms}; + } + + String buildFinishTask(String taskId) throws Exception { + Map message = Map.of( + "header", Map.of( + "action", "finish-task", + "task_id", taskId, + "streaming", "duplex"), + "payload", Map.of("input", Map.of())); + return objectMapper.writeValueAsString(message); + } + + /* ====================================================================== */ + /* WebSocket.Listener: collects events and signals task-started/finished. */ + /* ====================================================================== */ + + /** + * State machine for one DashScope ASR conversation. Package-private so + * unit tests can drive it with synthetic JSON without hitting the network. + */ + static class DashScopeSession implements WebSocket.Listener { + private final String taskId; + private final ObjectMapper mapper; + private final CountDownLatch taskStarted = new CountDownLatch(1); + private final CountDownLatch taskFinished = new CountDownLatch(1); + + /** + * Sentence buffer keyed by begin_time. DashScope emits multiple + * {@code result-generated} events for the same sentence as it gets + * refined (interim → final); each new event for a given begin_time + * supersedes the previous text. LinkedHashMap preserves arrival + * order, which roughly matches speech order, for the final concat. + */ + private final Map sentencesByBeginTime = new LinkedHashMap<>(); + + /** Buffer for fragmented text frames (WS allows partial messages). */ + private final StringBuilder textFrameBuf = new StringBuilder(); + + private final AtomicReference errorMessage = new AtomicReference<>(); + + /** Counts result-generated events — distinguishes "server got our audio + * but recognised nothing" (>0 events, all empty text) from "server + * saw zero audio frames" (0 events). Helps diagnose pacing / + * format issues. */ + private int resultEventCount; + + DashScopeSession(String taskId, ObjectMapper mapper) { + this.taskId = taskId; + this.mapper = mapper; + } + + @Override + public CompletionStage onText(WebSocket webSocket, CharSequence data, boolean last) { + textFrameBuf.append(data); + if (last) { + handleMessage(textFrameBuf.toString()); + textFrameBuf.setLength(0); + } + webSocket.request(1); + return null; + } + + @Override + public void onError(WebSocket webSocket, Throwable error) { + errorMessage.compareAndSet(null, "WS error: " + error.getMessage()); + taskStarted.countDown(); + taskFinished.countDown(); + } + + @Override + public CompletionStage onClose(WebSocket webSocket, int statusCode, String reason) { + // If the server closes before task-finished, unblock waiters. + if (taskFinished.getCount() > 0) { + errorMessage.compareAndSet(null, + "WS closed before task-finished (status=" + statusCode + ", reason=" + reason + ")"); + } + taskStarted.countDown(); + taskFinished.countDown(); + return null; + } + + /** Package-private hook so unit tests can drive {@link DashScopeSession} without a real WebSocket. */ + void handleMessage(String json) { + // Always trace the raw frame at DEBUG — this is invaluable when + // the protocol completes "successfully" but produces no + // transcripts. Without seeing every frame it's impossible to + // tell whether DashScope sent us a status-update / warning we + // ignored, or just stayed silent between task-started and + // task-finished. + log.debug("[DashScope STT] frame: {}", json); + try { + JsonNode node = mapper.readTree(json); + String event = node.path("header").path("event").asText(); + switch (event) { + case "task-started" -> taskStarted.countDown(); + case "result-generated" -> { + resultEventCount++; + JsonNode sentence = node.path("payload").path("output").path("sentence"); + if (sentence.isObject()) { + long beginTime = sentence.path("begin_time").asLong(0L); + String text = sentence.path("text").asText(""); + // Always overwrite — later events for the same begin_time + // carry the more-final transcript. + sentencesByBeginTime.put(beginTime, text); + } + } + case "task-finished" -> taskFinished.countDown(); + case "task-failed" -> { + String msg = node.path("header").path("error_message").asText("unknown"); + String code = node.path("header").path("error_code").asText(""); + errorMessage.compareAndSet(null, + code.isEmpty() ? msg : (code + " — " + msg)); + // Wake both latches so the caller can return the typed + // failure instead of timing out for the full budget. + taskStarted.countDown(); + taskFinished.countDown(); + } + // Anything else (status updates, model warnings, beta + // events) gets surfaced at INFO so it shows up without + // turning DEBUG on. If DashScope rolls out a new event + // type we should know about, this catches it. + default -> log.info("[DashScope STT] unhandled event '{}' frame={}", event, json); + } + } catch (Exception e) { + log.warn("[DashScope STT] failed to parse WS message: {}", e.getMessage()); + } + } + + boolean awaitTaskStarted(long timeout, TimeUnit unit) throws InterruptedException { + return taskStarted.await(timeout, unit); + } + + boolean awaitTaskFinished(long timeout, TimeUnit unit) throws InterruptedException { + return taskFinished.await(timeout, unit); + } + + boolean failed() { + return errorMessage.get() != null; + } + + String errorMessage() { + return errorMessage.get(); + } + + /** True once task-finished has been observed — used by the sender + * loop to bail out early instead of pacing through dead-WS sleeps. */ + boolean taskFinishedRaised() { + return taskFinished.getCount() == 0; + } + + int resultEventCount() { + return resultEventCount; + } + + String aggregatedText() { + // Concat in begin_time order. Different sentences typically don't + // need separator characters because Chinese text streams are + // already glued; for safety against missed punctuation we leave + // a soft join ("") rather than space — Whisper-style space + // joining produces odd-looking Chinese transcripts. + StringBuilder sb = new StringBuilder(); + sentencesByBeginTime.values().forEach(sb::append); + return sb.toString(); + } + } } diff --git a/mateclaw-server/src/main/java/vip/mate/stt/provider/OpenAiSttProvider.java b/mateclaw-server/src/main/java/vip/mate/stt/provider/OpenAiSttProvider.java index 923dd261..1ac7a75e 100644 --- a/mateclaw-server/src/main/java/vip/mate/stt/provider/OpenAiSttProvider.java +++ b/mateclaw-server/src/main/java/vip/mate/stt/provider/OpenAiSttProvider.java @@ -8,6 +8,7 @@ import lombok.RequiredArgsConstructor; import lombok.extern.slf4j.Slf4j; import org.springframework.stereotype.Component; import vip.mate.llm.service.ModelProviderService; +import vip.mate.stt.AudioMimeTypes; import vip.mate.stt.SttProvider; import vip.mate.stt.SttRequest; import vip.mate.stt.SttResult; @@ -33,10 +34,33 @@ public class OpenAiSttProvider implements SttProvider { @Override public boolean requiresCredential() { return true; } @Override public int autoDetectOrder() { return 100; } + /** + * Whisper is the canonical English STT and noticeably weaker on Chinese + * (it tends to produce simplified-character output even for traditional + * input, and short Chinese clips frequently transcribe to gibberish). + * Boost Whisper's priority for English/Japanese/Korean (where it leads), + * and de-prioritise it for Chinese so DashScope (Paraformer) wins the + * auto-pick. + */ + @Override + public int autoDetectOrder(String language) { + if (language == null) return autoDetectOrder(); + String lang = language.toLowerCase(); + if (lang.startsWith("zh")) return 250; // pushed below DashScope Paraformer + if (lang.startsWith("en") + || lang.startsWith("ja") + || lang.startsWith("ko")) return 80; // pulled above DashScope + return autoDetectOrder(); + } + @Override public boolean isAvailable(SystemSettingsDTO config) { - try { return modelProviderService.isProviderConfigured("openai"); } - catch (Exception e) { return false; } + try { + return modelProviderService.isProviderConfigured("openai"); + } catch (Exception e) { + log.warn("[OpenAI STT] availability check failed: {}", e.getMessage()); + return false; + } } @Override @@ -48,12 +72,18 @@ public class OpenAiSttProvider implements SttProvider { String url = (baseUrl != null ? baseUrl : "https://api.openai.com") + "/v1/audio/transcriptions"; String model = request.getModel() != null ? request.getModel() : DEFAULT_MODEL; - String fileName = request.getFileName() != null ? request.getFileName() : "audio.ogg"; + // AudioMimeTypes ensures the filename extension matches the + // actual bytes (audio.wav, audio.mp3, etc.), which Hutool then + // uses to infer the multipart Content-Type. Don't pass + // contentType to .form() explicitly — Hutool has no + // form(String,byte[],String,String) overload, and the wrong + // dispatch crashes with ClassCastException on byte[] → Object[]. + String fileName = AudioMimeTypes.resolveFileName(request.getFileName(), request.getContentType()); HttpResponse response = HttpRequest.post(url) .header("Authorization", "Bearer " + apiKey) .form("model", model) - .form("file", request.getAudioData(), request.getContentType(), fileName) + .form("file", request.getAudioData(), fileName) .timeout(60_000) .execute(); diff --git a/mateclaw-server/src/main/resources/db/data-en.sql b/mateclaw-server/src/main/resources/db/data-en.sql index ab14a78f..fa383730 100644 --- a/mateclaw-server/src/main/resources/db/data-en.sql +++ b/mateclaw-server/src/main/resources/db/data-en.sql @@ -329,6 +329,22 @@ MERGE INTO mate_system_setting (id, setting_key, setting_value, description, cre KEY (id) VALUES (1000000013, 'searxngBaseUrl', '', 'SearXNG instance base URL (auto-configured in Docker)', NOW(), NOW()); +-- Speech-to-text (STT) defaults — enabled out of the box so users only need to configure an API key. +-- Skip-if-exists on setting_key (NOT MERGE BY id) so we don't trip the +-- UNIQUE index when an existing user has the row at a runtime-assigned id +-- from toggling the UI before this seed shipped. V46 covers the same. +INSERT INTO mate_system_setting (id, setting_key, setting_value, description, create_time, update_time) +SELECT 1000000020, 'sttEnabled', 'true', 'Enable speech-to-text (TalkMode mic input)', NOW(), NOW() +WHERE NOT EXISTS (SELECT 1 FROM mate_system_setting WHERE setting_key = 'sttEnabled'); + +INSERT INTO mate_system_setting (id, setting_key, setting_value, description, create_time, update_time) +SELECT 1000000021, 'sttProvider', 'auto', 'STT provider: auto / openai / dashscope', NOW(), NOW() +WHERE NOT EXISTS (SELECT 1 FROM mate_system_setting WHERE setting_key = 'sttProvider'); + +INSERT INTO mate_system_setting (id, setting_key, setting_value, description, create_time, update_time) +SELECT 1000000022, 'sttFallbackEnabled', 'true', 'Try alternate STT provider when the primary fails', NOW(), NOW() +WHERE NOT EXISTS (SELECT 1 FROM mate_system_setting WHERE setting_key = 'sttFallbackEnabled'); + -- Built-in tool: Date & Time MERGE INTO mate_tool (id, name, display_name, description, tool_type, bean_name, icon, enabled, builtin, create_time, update_time, deleted) KEY (id) diff --git a/mateclaw-server/src/main/resources/db/data-mysql-en.sql b/mateclaw-server/src/main/resources/db/data-mysql-en.sql index 32ebf72d..3ddd52ab 100644 --- a/mateclaw-server/src/main/resources/db/data-mysql-en.sql +++ b/mateclaw-server/src/main/resources/db/data-mysql-en.sql @@ -368,6 +368,26 @@ INSERT INTO mate_system_setting (id, setting_key, setting_value, description, cr VALUES (1000000013, 'searxngBaseUrl', '', 'SearXNG instance base URL (auto-configured in Docker)', NOW(), NOW()) ON DUPLICATE KEY UPDATE setting_key=VALUES(setting_key), setting_value=VALUES(setting_value), description=VALUES(description), update_time=VALUES(update_time); +-- Speech-to-text (STT) defaults — enabled out of the box so users only need to configure an API key. +-- Skip-if-exists keyed on setting_key (FROM DUAL ... WHERE NOT EXISTS) so +-- we don't override a value the user explicitly set before this seed shipped, +-- and don't trip the UNIQUE index on setting_key when their row is at a +-- runtime-assigned id. V46 migration uses the same idiom. +INSERT INTO mate_system_setting (id, setting_key, setting_value, description, create_time, update_time) +SELECT 1000000020, 'sttEnabled', 'true', 'Enable speech-to-text (TalkMode mic input)', NOW(), NOW() +FROM DUAL +WHERE NOT EXISTS (SELECT 1 FROM mate_system_setting WHERE setting_key = 'sttEnabled'); + +INSERT INTO mate_system_setting (id, setting_key, setting_value, description, create_time, update_time) +SELECT 1000000021, 'sttProvider', 'auto', 'STT provider: auto / openai / dashscope', NOW(), NOW() +FROM DUAL +WHERE NOT EXISTS (SELECT 1 FROM mate_system_setting WHERE setting_key = 'sttProvider'); + +INSERT INTO mate_system_setting (id, setting_key, setting_value, description, create_time, update_time) +SELECT 1000000022, 'sttFallbackEnabled', 'true', 'Try alternate STT provider when the primary fails', NOW(), NOW() +FROM DUAL +WHERE NOT EXISTS (SELECT 1 FROM mate_system_setting WHERE setting_key = 'sttFallbackEnabled'); + -- Built-in tool: Date & Time INSERT INTO mate_tool (id, name, display_name, description, tool_type, bean_name, icon, enabled, builtin, create_time, update_time, deleted) VALUES (1000000001, 'DateTimeTool', 'Date & Time', 'Get current date and time information', 'builtin', 'dateTimeTool', '🕐', TRUE, TRUE, NOW(), NOW(), 0) diff --git a/mateclaw-server/src/main/resources/db/data-mysql-zh.sql b/mateclaw-server/src/main/resources/db/data-mysql-zh.sql index 4c50152f..94afd8e3 100644 --- a/mateclaw-server/src/main/resources/db/data-mysql-zh.sql +++ b/mateclaw-server/src/main/resources/db/data-mysql-zh.sql @@ -368,6 +368,25 @@ INSERT INTO mate_system_setting (id, setting_key, setting_value, description, cr VALUES (1000000013, 'searxngBaseUrl', '', 'SearXNG 实例地址(Docker 部署时自动配置)', NOW(), NOW()) ON DUPLICATE KEY UPDATE setting_key=VALUES(setting_key), setting_value=VALUES(setting_value), description=VALUES(description), update_time=VALUES(update_time); +-- 语音识别(STT)默认配置 —— 默认启用,用户只需在模型管理中配置 OpenAI / DashScope API Key 即可使用 +-- 用 setting_key 的 skip-if-exists 写法(FROM DUAL ... WHERE NOT EXISTS), +-- 既不强行覆盖用户显式设过的值,也不会撞 setting_key UNIQUE 索引。 +-- V46 迁移走的是同一套语义。 +INSERT INTO mate_system_setting (id, setting_key, setting_value, description, create_time, update_time) +SELECT 1000000020, 'sttEnabled', 'true', '启用语音识别(TalkMode 麦克风输入)', NOW(), NOW() +FROM DUAL +WHERE NOT EXISTS (SELECT 1 FROM mate_system_setting WHERE setting_key = 'sttEnabled'); + +INSERT INTO mate_system_setting (id, setting_key, setting_value, description, create_time, update_time) +SELECT 1000000021, 'sttProvider', 'auto', 'STT 提供商:auto / openai / dashscope', NOW(), NOW() +FROM DUAL +WHERE NOT EXISTS (SELECT 1 FROM mate_system_setting WHERE setting_key = 'sttProvider'); + +INSERT INTO mate_system_setting (id, setting_key, setting_value, description, create_time, update_time) +SELECT 1000000022, 'sttFallbackEnabled', 'true', '主 provider 失败时自动尝试备选 provider', NOW(), NOW() +FROM DUAL +WHERE NOT EXISTS (SELECT 1 FROM mate_system_setting WHERE setting_key = 'sttFallbackEnabled'); + -- 内置工具:日期时间 INSERT INTO mate_tool (id, name, display_name, description, tool_type, bean_name, icon, enabled, builtin, create_time, update_time, deleted) VALUES (1000000001, 'DateTimeTool', '日期时间', '获取当前日期和时间信息', 'builtin', 'dateTimeTool', '🕐', TRUE, TRUE, NOW(), NOW(), 0) diff --git a/mateclaw-server/src/main/resources/db/data-zh.sql b/mateclaw-server/src/main/resources/db/data-zh.sql index 9c75e089..e5d13c91 100644 --- a/mateclaw-server/src/main/resources/db/data-zh.sql +++ b/mateclaw-server/src/main/resources/db/data-zh.sql @@ -333,6 +333,22 @@ MERGE INTO mate_system_setting (id, setting_key, setting_value, description, cre KEY (id) VALUES (1000000013, 'searxngBaseUrl', '', 'SearXNG 实例地址(Docker 部署时自动配置)', NOW(), NOW()); +-- 语音识别(STT)默认配置 —— 默认启用,用户只需在模型管理中配置 OpenAI / DashScope API Key 即可使用 +-- 用 setting_key 的 skip-if-exists 写法(不再走 MERGE BY id),避免与 +-- 旧版 UI 已经写入的运行时 snowflake id 在 setting_key UNIQUE 索引上撞车。 +-- V46 迁移走的是同一套语义。 +INSERT INTO mate_system_setting (id, setting_key, setting_value, description, create_time, update_time) +SELECT 1000000020, 'sttEnabled', 'true', '启用语音识别(TalkMode 麦克风输入)', NOW(), NOW() +WHERE NOT EXISTS (SELECT 1 FROM mate_system_setting WHERE setting_key = 'sttEnabled'); + +INSERT INTO mate_system_setting (id, setting_key, setting_value, description, create_time, update_time) +SELECT 1000000021, 'sttProvider', 'auto', 'STT 提供商:auto / openai / dashscope', NOW(), NOW() +WHERE NOT EXISTS (SELECT 1 FROM mate_system_setting WHERE setting_key = 'sttProvider'); + +INSERT INTO mate_system_setting (id, setting_key, setting_value, description, create_time, update_time) +SELECT 1000000022, 'sttFallbackEnabled', 'true', '主 provider 失败时自动尝试备选 provider', NOW(), NOW() +WHERE NOT EXISTS (SELECT 1 FROM mate_system_setting WHERE setting_key = 'sttFallbackEnabled'); + -- 内置工具:日期时间 MERGE INTO mate_tool (id, name, display_name, description, tool_type, bean_name, icon, enabled, builtin, create_time, update_time, deleted) KEY (id) diff --git a/mateclaw-server/src/main/resources/db/migration/h2/V46__stt_default_enabled.sql b/mateclaw-server/src/main/resources/db/migration/h2/V46__stt_default_enabled.sql new file mode 100644 index 00000000..d179dc97 --- /dev/null +++ b/mateclaw-server/src/main/resources/db/migration/h2/V46__stt_default_enabled.sql @@ -0,0 +1,31 @@ +-- Default-enable STT on existing deployments. New installs run V46 first +-- (table empty, INSERT succeeds), then DatabaseBootstrapRunner re-runs the +-- same statements from data-{en,zh,mysql-en,mysql-zh}.sql with the same +-- skip-if-exists semantics, so this is idempotent across both paths. +-- +-- Why default to enabled: STT requires a recording UI gesture to even be +-- exercised, so leaving it off by default just makes "the mic button does +-- nothing" the most-reported support issue. Once enabled, it still does +-- nothing dangerous unless the user has also configured an OpenAI / DashScope +-- API key — those are the real gating credentials. +-- +-- Idiom: INSERT ... SELECT ... WHERE NOT EXISTS, keyed on setting_key. +-- Earlier versions of this migration used MERGE INTO ... KEY (id), which +-- crashed on deployments where the user had already toggled STT in the UI: +-- their row landed at a runtime-assigned snowflake id, then this migration +-- tried to insert a fresh row at id=1000000020 and tripped the UNIQUE index +-- on setting_key. Skip-if-exists preserves whatever value the user picked +-- (don't override an explicit "off" with "on"). See +-- https://git.mate.vip/mate/MateClaw issue noted 2026-04-26. + +INSERT INTO mate_system_setting (id, setting_key, setting_value, description, create_time, update_time) +SELECT 1000000020, 'sttEnabled', 'true', 'Enable speech-to-text (TalkMode mic input)', NOW(), NOW() +WHERE NOT EXISTS (SELECT 1 FROM mate_system_setting WHERE setting_key = 'sttEnabled'); + +INSERT INTO mate_system_setting (id, setting_key, setting_value, description, create_time, update_time) +SELECT 1000000021, 'sttProvider', 'auto', 'STT provider: auto / openai / dashscope', NOW(), NOW() +WHERE NOT EXISTS (SELECT 1 FROM mate_system_setting WHERE setting_key = 'sttProvider'); + +INSERT INTO mate_system_setting (id, setting_key, setting_value, description, create_time, update_time) +SELECT 1000000022, 'sttFallbackEnabled', 'true', 'Try alternate STT provider when the primary fails', NOW(), NOW() +WHERE NOT EXISTS (SELECT 1 FROM mate_system_setting WHERE setting_key = 'sttFallbackEnabled'); diff --git a/mateclaw-server/src/main/resources/db/migration/mysql/V46__stt_default_enabled.sql b/mateclaw-server/src/main/resources/db/migration/mysql/V46__stt_default_enabled.sql new file mode 100644 index 00000000..3f14519f --- /dev/null +++ b/mateclaw-server/src/main/resources/db/migration/mysql/V46__stt_default_enabled.sql @@ -0,0 +1,24 @@ +-- Default-enable STT on existing deployments. See the h2/ counterpart for +-- the "why enabled by default" rationale and the bug history that drove +-- the skip-if-exists idiom. Same V number is used in h2/ for cross-dialect +-- parity. +-- +-- MySQL doesn't allow `INSERT ... SELECT ... WHERE NOT EXISTS` without a +-- FROM clause, so we synthesise one with `FROM DUAL`. The end result is +-- the same: insert when the setting_key is absent, no-op when it's already +-- there. + +INSERT INTO mate_system_setting (id, setting_key, setting_value, description, create_time, update_time) +SELECT 1000000020, 'sttEnabled', 'true', 'Enable speech-to-text (TalkMode mic input)', NOW(), NOW() +FROM DUAL +WHERE NOT EXISTS (SELECT 1 FROM mate_system_setting WHERE setting_key = 'sttEnabled'); + +INSERT INTO mate_system_setting (id, setting_key, setting_value, description, create_time, update_time) +SELECT 1000000021, 'sttProvider', 'auto', 'STT provider: auto / openai / dashscope', NOW(), NOW() +FROM DUAL +WHERE NOT EXISTS (SELECT 1 FROM mate_system_setting WHERE setting_key = 'sttProvider'); + +INSERT INTO mate_system_setting (id, setting_key, setting_value, description, create_time, update_time) +SELECT 1000000022, 'sttFallbackEnabled', 'true', 'Try alternate STT provider when the primary fails', NOW(), NOW() +FROM DUAL +WHERE NOT EXISTS (SELECT 1 FROM mate_system_setting WHERE setting_key = 'sttFallbackEnabled'); diff --git a/mateclaw-ui/src/components/chat/TalkMode.vue b/mateclaw-ui/src/components/chat/TalkMode.vue index c40fb634..66696d04 100644 --- a/mateclaw-ui/src/components/chat/TalkMode.vue +++ b/mateclaw-ui/src/components/chat/TalkMode.vue @@ -37,13 +37,13 @@ @@ -55,6 +55,7 @@ import { ref, computed, onMounted, onBeforeUnmount } from 'vue' import { useI18n } from 'vue-i18n' import { ElMessage } from 'element-plus' import { CloseBold, Loading, Microphone, Service } from '@element-plus/icons-vue' +import { WavRecorder } from '@/utils/wavEncoder' const { t } = useI18n() @@ -68,23 +69,53 @@ const emit = defineEmits<{ close: [] }>() -type TalkState = 'idle' | 'listening' | 'processing' | 'speaking' +/** + * Connection-aware state machine. Pre-fix the modal opened in 'idle' which + * lit the PTT button green even while the WebSocket was still in + * CONNECTING (readyState=0). Users hit the button immediately, recorded a + * clip, and stopListening saw ws.readyState !== OPEN — the audio went into + * the void. Now PTT is disabled until the backend sends {type:"ready"}. + */ +type TalkState = 'connecting' | 'idle' | 'listening' | 'processing' | 'speaking' | 'failed' -const state = ref('idle') +const state = ref('connecting') const transcript = ref>([]) let ws: WebSocket | null = null -let mediaRecorder: MediaRecorder | null = null -let audioChunks: Blob[] = [] +let recorder: WavRecorder | null = null +/** + * Persistent warmed-up recorder kept alive for the modal's lifetime so the + * press-and-hold gesture doesn't race a first-time mic permission dialog. + * The dialog steals focus → mouseup fires on the dialog instead of the PTT + * button → stopListening never runs → recording is stuck on. Warming up + * front-loads the permission prompt and reuses the MediaStream. + */ +let warmRecorder: WavRecorder | null = null let audioContext: AudioContext | null = null +/** + * Button-enabled predicate. Allows: + * - 'idle' — fresh / between recordings (the normal case) + * - 'listening' — already recording, the release event must reach us + * - 'failed' — clicking restarts the WS connection (retry path) + */ +const canRecord = computed(() => + state.value === 'idle' || state.value === 'listening' || state.value === 'failed') +const pttLabel = computed(() => { + if (state.value === 'connecting') return t('talk.connecting') + if (state.value === 'failed') return t('talk.retry') + return state.value === 'listening' ? t('talk.releaseToSend') : t('talk.holdToTalk') +}) + const stateClass = computed(() => 'talk-state--' + state.value) const stateLabel = computed(() => { switch (state.value) { + case 'connecting': return t('talk.connecting') case 'idle': return t('talk.ready') case 'listening': return t('talk.listening') case 'processing': return t('talk.processing') case 'speaking': return t('talk.speaking') + case 'failed': return t('talk.connectionError') default: return '' } }) @@ -92,11 +123,20 @@ const stateLabel = computed(() => { onMounted(() => { if (props.agentId) { connectWebSocket() + // Pre-acquire mic permission so the press-and-hold path skips the + // permission dialog. Best-effort — failure here just means the user + // gets the dialog on first PTT press (i.e. previous behaviour). + warmRecorder = new WavRecorder() + warmRecorder.warmUp().catch(err => { + console.debug('[TalkMode] mic warm-up failed (will prompt on first PTT)', err) + }) } }) onBeforeUnmount(() => { disconnectWebSocket() + warmRecorder?.releaseWarmUp() + warmRecorder = null }) function connectWebSocket() { @@ -104,10 +144,13 @@ function connectWebSocket() { const token = localStorage.getItem('token') const wsUrl = `${protocol}//${location.host}/api/v1/talk/ws${token ? '?token=' + token : ''}` + state.value = 'connecting' ws = new WebSocket(wsUrl) ws.onopen = () => { - // Send init message + // Send init message. Stay in 'connecting' until the backend's + // {type:"ready"} ack lands — the WS being open isn't the same as the + // backend session being initialised. ws?.send(JSON.stringify({ type: 'init', agentId: props.agentId, @@ -127,6 +170,7 @@ function connectWebSocket() { const data = JSON.parse(event.data) switch (data.type) { case 'ready': + // Only NOW does the PTT button unlock — see canRecord computed. state.value = 'idle' break case 'state': @@ -151,14 +195,40 @@ function connectWebSocket() { } } - ws.onclose = () => { - state.value = 'idle' + ws.onerror = (e) => { + console.warn('[TalkMode] WS error', e) + if (state.value === 'connecting') { + state.value = 'failed' + ElMessage.error(t('talk.connectionError')) + } + } + + ws.onclose = (e) => { + console.debug('[TalkMode] WS closed', 'code=', e.code, 'reason=', e.reason) + // Distinguish close-before-init (the user got a failed handshake) from + // close-after-success (the modal was just closed). A premature close + // leaves the user staring at "Ready" with no working button — flag + // it so the retry path actually runs. + if (state.value === 'connecting') { + state.value = 'failed' + } else if (state.value !== 'failed') { + state.value = 'idle' + } } } +/** Manual reconnect — wired to the PTT button when state==='failed'. */ +function retryConnection() { + ws?.close() + ws = null + connectWebSocket() +} + function disconnectWebSocket() { - if (mediaRecorder && mediaRecorder.state !== 'inactive') { - mediaRecorder.stop() + // Force-stop any in-flight recording so released the mic on close. + if (recorder?.isActive()) { + recorder.stop().catch(() => {}) + recorder = null } ws?.close() ws = null @@ -167,50 +237,74 @@ function disconnectWebSocket() { } async function startListening() { + // Failed-state click is a retry, not a recording start. The button label + // already says "Retry" via pttLabel, so this matches the user's intent. + if (state.value === 'failed') { + retryConnection() + return + } if (state.value !== 'idle') return try { - const stream = await navigator.mediaDevices.getUserMedia({ audio: true }) - audioChunks = [] - mediaRecorder = new MediaRecorder(stream, { mimeType: 'audio/webm;codecs=opus' }) - - mediaRecorder.ondataavailable = (event) => { - if (event.data.size > 0) { - audioChunks.push(event.data) - } - } - - mediaRecorder.onstop = async () => { - // Stop all tracks - stream.getTracks().forEach(t => t.stop()) - - if (audioChunks.length === 0) { - state.value = 'idle' - return - } - - // Combine chunks and send - const audioBlob = new Blob(audioChunks, { type: 'audio/webm' }) - const arrayBuffer = await audioBlob.arrayBuffer() - if (ws && ws.readyState === WebSocket.OPEN) { - ws.send(arrayBuffer) - state.value = 'processing' - } else { - state.value = 'idle' - } - } - - mediaRecorder.start() + // Web Audio API + manual WAV encode (utils/wavEncoder.ts) — replaces + // MediaRecorder/WebM. DashScope Paraformer rejects webm; WAV is the + // lowest common denominator every STT provider accepts. + // + // Reuse the warmed-up recorder so we skip the permission dialog if + // it was successfully acquired in onMounted. If warm-up failed (or + // is still pending), fall back to a fresh recorder. + recorder = warmRecorder ?? new WavRecorder() + warmRecorder = null // ownership transferred for the duration of recording + await recorder.start() state.value = 'listening' - } catch { + console.debug('[TalkMode] listening started') + } catch (err) { + console.warn('[TalkMode] startListening failed', err) ElMessage.error(t('talk.micError')) state.value = 'idle' + recorder = null } } -function stopListening() { - if (state.value !== 'listening' || !mediaRecorder) return - mediaRecorder.stop() +async function stopListening() { + // Belt-and-braces: also stop when recorder exists but state hasn't caught + // up (race with the slow path of startListening). Without this, a fast + // press-then-release leaves the recorder running invisibly. + if (!recorder) { + if (state.value === 'listening') state.value = 'idle' + return + } + const result = await recorder.stop() + recorder = null + // Re-warm for the next press so subsequent PTTs also skip permission. + warmRecorder = new WavRecorder() + warmRecorder.warmUp().catch(() => {}) + + if (!result) { + // No audio captured — the most common cause is ScriptProcessor never + // firing (suspended AudioContext) or the recording was so short no + // sample buffer landed. Surface a clear hint instead of a silent idle. + console.warn('[TalkMode] stop returned no audio (recording too short or context suspended)') + ElMessage.warning(t('talk.tooShort') || '录音过短,请按住按钮多说几秒') + state.value = 'idle' + return + } + const arrayBuffer = await result.blob.arrayBuffer() + console.debug('[TalkMode] stop produced wav', + 'bytes=', arrayBuffer.byteLength, + 'duration=', result.durationSeconds, 's', + 'wsReadyState=', ws?.readyState) + if (ws && ws.readyState === WebSocket.OPEN) { + ws.send(arrayBuffer) + state.value = 'processing' + } else { + // Distinct from the empty-recording case: capture worked, but the WS + // dropped before we got here. Tell the user instead of silently going + // idle — they'd otherwise blame the mic. + console.warn('[TalkMode] WS not open at stop time, readyState=', ws?.readyState) + ElMessage.error(t('talk.connectionError')) + state.value = 'idle' + } } async function playAudio(blob: Blob) { diff --git a/mateclaw-ui/src/i18n/locales/en-US.ts b/mateclaw-ui/src/i18n/locales/en-US.ts index c1731cc3..4ee983e7 100644 --- a/mateclaw-ui/src/i18n/locales/en-US.ts +++ b/mateclaw-ui/src/i18n/locales/en-US.ts @@ -485,7 +485,7 @@ export default { sttProvider: 'Select preferred STT provider. Auto mode picks the first available one.', sttFallbackEnabled: 'Automatically try other configured providers if the preferred one fails.', openaiSttInfo: 'Reuses OpenAI API Key from Model Management. Whisper model, supports multilingual auto-detection.', - dashscopeSttInfo: 'Reuses DashScope API Key from Model Management. Paraformer model, excellent Chinese recognition.', + dashscopeSttInfo: 'Reuses DashScope API Key from Model Management. Paraformer Realtime over WebSocket — strong Chinese recognition, sub-second latency.', musicEnabled: 'Enable to let Agent use the music generation tool. Google Lyria reuses Google Key.', musicProvider: 'Select preferred music provider. Auto mode prioritizes Google Lyria.', musicFallbackEnabled: 'Automatically try other configured providers if the preferred one fails.', @@ -1817,16 +1817,19 @@ export default { talk: { title: 'Talk Mode', ready: 'Ready', + connecting: 'Connecting...', listening: 'Listening...', processing: 'Thinking...', speaking: 'Speaking...', holdToTalk: 'Hold to talk', releaseToSend: 'Release to send', + retry: 'Tap to retry', you: 'You', ai: 'AI', micError: 'Cannot access microphone', - connectionError: 'Connection error', + connectionError: 'Connection error — tap the button to retry', playbackError: 'Audio playback failed', + tooShort: 'Recording too short — hold the button while speaking', }, browser: { timeline: { diff --git a/mateclaw-ui/src/i18n/locales/zh-CN.ts b/mateclaw-ui/src/i18n/locales/zh-CN.ts index 72553ab2..c10a88cd 100644 --- a/mateclaw-ui/src/i18n/locales/zh-CN.ts +++ b/mateclaw-ui/src/i18n/locales/zh-CN.ts @@ -481,7 +481,7 @@ export default { sttProvider: '选择首选 STT Provider,auto 模式自动选择可用的 Provider。', sttFallbackEnabled: '首选 Provider 失败时自动尝试其他已配置的 Provider。', openaiSttInfo: '复用模型管理中的 OpenAI API Key。使用 Whisper 模型,支持多语言自动识别。', - dashscopeSttInfo: '复用模型管理中的 DashScope API Key。使用 Paraformer 模型,中文识别效果优秀。', + dashscopeSttInfo: '复用模型管理中的 DashScope API Key。使用 Paraformer Realtime(WebSocket 流式),中文识别效果优秀,亚秒级延迟。', // 音乐生成 musicEnabled: '开启后 Agent 可通过 music_generate 工具生成音乐。Google Lyria 复用 Google Key。', musicProvider: '选择首选音乐 Provider,auto 模式优先使用 Google Lyria。', @@ -1827,16 +1827,19 @@ export default { talk: { title: '语音模式', ready: '就绪', + connecting: '连接中...', listening: '收听中...', processing: '思考中...', speaking: '回复中...', holdToTalk: '按住说话', releaseToSend: '松开发送', + retry: '点击重新连接', you: '你', ai: 'AI', micError: '无法访问麦克风', - connectionError: '连接错误', + connectionError: '连接错误,点击按钮重试', playbackError: '音频播放失败', + tooShort: '录音过短,请按住按钮多说几秒', }, browser: { timeline: { diff --git a/mateclaw-ui/src/utils/wavEncoder.ts b/mateclaw-ui/src/utils/wavEncoder.ts new file mode 100644 index 00000000..de5fb383 --- /dev/null +++ b/mateclaw-ui/src/utils/wavEncoder.ts @@ -0,0 +1,277 @@ +/** + * Browser-side recorder that captures microphone audio via the Web Audio API + * and encodes it directly to a 16-bit PCM WAV blob. + * + *

Why this exists: MediaRecorder produces WebM/Opus, which DashScope + * Paraformer rejects (it accepts wav/mp3/m4a/flac/aac/amr/ogg-Vorbis only). + * OpenAI Whisper claims WebM support but is finicky with the codecs string + * MediaRecorder picks. WAV is the lowest common denominator that every STT + * provider accepts without server-side transcoding. + * + *

Trade-off: WAV files are ~10x larger than Opus. For typical + * conversational STT (5-30s clips at 16 kHz / 16-bit / mono) that's + * 160 KB - 1 MB — fine to send over a websocket. + * + *

Reference: openclaw {@code src/media/audio-transcode.ts} solves the same + * problem with server-side ffmpeg. Doing it client-side avoids the ffmpeg + * dependency on the MateClaw server and works the same way. + */ + +/** Target sample rate for recordings. 16 kHz is the standard for STT models. */ +const TARGET_SAMPLE_RATE = 16_000; + +/** WAV / PCM constants. */ +const NUM_CHANNELS = 1; +const BITS_PER_SAMPLE = 16; +const RIFF_HEADER_SIZE = 44; + +/** + * Result of a finished recording. + */ +export interface WavRecording { + /** WAV-encoded audio bytes ready to upload. */ + blob: Blob; + /** Convenience MIME type — always "audio/wav". */ + mimeType: 'audio/wav'; + /** Recording duration in seconds (rounded to ms). */ + durationSeconds: number; +} + +/** + * Live microphone recorder backed by Web Audio API. Call {@link start} once, + * later call {@link stop} to receive a {@link WavRecording}. After stop the + * underlying MediaStream tracks are released. + * + *

The recorder captures Float32 samples from a {@code ScriptProcessorNode}, + * accumulates them, then encodes a 16-bit PCM mono WAV in {@link stop}. + * ScriptProcessorNode is deprecated in favour of AudioWorklet, but it's still + * universally supported and zero-config — sufficient for short STT clips. + * + *

Browser support: every browser MateClaw targets ships AudioContext + + * ScriptProcessorNode. Safari requires a user gesture to call + * {@code getUserMedia}; the caller should already be doing this. + */ +export class WavRecorder { + private audioContext: AudioContext | null = null; + private mediaStream: MediaStream | null = null; + private source: MediaStreamAudioSourceNode | null = null; + private processor: ScriptProcessorNode | null = null; + /** Silent sink node — keeps the processor graph alive without echoing through speakers. */ + private silentSink: GainNode | null = null; + private chunks: Float32Array[] = []; + /** Sample rate the AudioContext is actually running at — may differ from TARGET_SAMPLE_RATE. */ + private inputSampleRate = TARGET_SAMPLE_RATE; + private startTimeMs = 0; + + /** + * Pre-acquire microphone permission and warm the audio graph without + * starting capture. Called when the TalkMode modal opens so the + * press-and-hold gesture later doesn't race a first-time permission + * dialog (the dialog steals focus → mouseup fires on the dialog → + * stopListening never runs → recording is stuck on forever). + * + *

Safe to call multiple times. Subsequent calls return immediately. + */ + async warmUp(): Promise { + if (this.mediaStream) return; + this.mediaStream = await navigator.mediaDevices.getUserMedia({ audio: true }); + } + + /** + * Begin capturing from the microphone. Throws if mic access is denied. + * Already-started recorders are idempotent — calling start twice is a no-op. + */ + async start(): Promise { + if (this.audioContext) return; + + // sampleRate hint: browsers honour it on Chrome/Edge but Safari may + // ignore it and run at the device default. We resample manually in + // stop() to be safe. + const ctx = new AudioContext(); + this.audioContext = ctx; + this.inputSampleRate = ctx.sampleRate; + // Reuse the warmed-up stream when present so we skip the permission + // dialog on the press-and-hold path. + if (!this.mediaStream) { + this.mediaStream = await navigator.mediaDevices.getUserMedia({ audio: true }); + } + // Modern Chromium AudioContexts created from a user gesture may still + // arrive suspended if mic permission was prompted asynchronously. + // Force-resume so onaudioprocess actually fires. + if (ctx.state === 'suspended') { + await ctx.resume(); + } + this.source = ctx.createMediaStreamSource(this.mediaStream); + this.processor = ctx.createScriptProcessor(4096, 1, 1); + this.processor.onaudioprocess = (e) => { + const channel = e.inputBuffer.getChannelData(0); + // Defensive copy — the underlying buffer is reused on the next callback. + this.chunks.push(new Float32Array(channel)); + }; + // Wire source → processor → silent gain → destination. The gain=0 + // node muzzles the echo through the speakers but keeps the chain + // attached to destination, which Chrome requires to fire + // onaudioprocess. Connecting processor directly to destination + // would echo the mic input back through speakers (feedback) AND + // some browsers stop calling onaudioprocess if they decide the + // chain "produces no audible output" — the explicit GainNode + // makes that decision unambiguous. + const silentSink = ctx.createGain(); + silentSink.gain.value = 0; + this.silentSink = silentSink; + this.source.connect(this.processor); + this.processor.connect(silentSink); + silentSink.connect(ctx.destination); + this.startTimeMs = Date.now(); + // Diagnostic — paste from devtools console when the recording silently + // produces 0 bytes. Includes sample rate so we can confirm Safari + // is at 44.1kHz vs Chrome's 48kHz. + console.debug('[WavRecorder] started', + 'sampleRate=', ctx.sampleRate, + 'state=', ctx.state); + } + + /** + * Stop capturing, encode the buffered audio to WAV, and release resources. + * Returns null when nothing was captured (e.g. start failed silently). + */ + async stop(): Promise { + if (!this.audioContext) return null; + + const durationSeconds = Math.round((Date.now() - this.startTimeMs)) / 1000; + const sampleRate = this.inputSampleRate; + const collected = this.chunks; + + this.processor?.disconnect(); + this.source?.disconnect(); + this.silentSink?.disconnect(); + this.mediaStream?.getTracks().forEach((t) => t.stop()); + await this.audioContext.close(); + this.audioContext = null; + this.mediaStream = null; + this.source = null; + this.processor = null; + this.silentSink = null; + this.chunks = []; + + // Diagnostic — chunk count + total samples. If chunks=0, the + // ScriptProcessor never fired (suspended context, browser + // optimised the graph away, etc). With this log the user can paste + // the devtools output and we can tell from a single line. + const totalSamples = collected.reduce((n, c) => n + c.length, 0); + console.debug('[WavRecorder] stopped', + 'durationSec=', durationSeconds, + 'chunks=', collected.length, + 'totalSamples=', totalSamples); + + if (collected.length === 0) { + return null; + } + const merged = mergeFloat32(collected); + // Resample to 16 kHz if we captured at a higher rate (Safari often runs + // the AudioContext at 44.1 kHz). 16 kHz is what Whisper / Paraformer + // expect, and shrinks the WAV by ~3x at no quality cost for speech. + const resampled = sampleRate === TARGET_SAMPLE_RATE + ? merged + : downsample(merged, sampleRate, TARGET_SAMPLE_RATE); + const blob = encodeWav(resampled, TARGET_SAMPLE_RATE); + return { blob, mimeType: 'audio/wav', durationSeconds }; + } + + /** True when the recorder is actively capturing. */ + isActive(): boolean { + return this.audioContext !== null; + } + + /** + * Release the warmed-up MediaStream without starting a recording. + * Pair with {@link warmUp} on modal close. + */ + releaseWarmUp(): void { + if (this.audioContext) return; // active recording owns the stream + this.mediaStream?.getTracks().forEach((t) => t.stop()); + this.mediaStream = null; + } +} + +/* ------------------------------------------------------------------ */ +/* Internals */ +/* ------------------------------------------------------------------ */ + +function mergeFloat32(chunks: Float32Array[]): Float32Array { + let total = 0; + for (const c of chunks) total += c.length; + const out = new Float32Array(total); + let offset = 0; + for (const c of chunks) { + out.set(c, offset); + offset += c.length; + } + return out; +} + +/** + * Decimating linear-interpolation resampler. Adequate for 16 kHz speech — + * the accuracy gap vs polyphase resampling is inaudible to Whisper / + * Paraformer at the input sample rates we see in practice (44.1k → 16k). + */ +function downsample(samples: Float32Array, fromRate: number, toRate: number): Float32Array { + if (fromRate === toRate) return samples; + const ratio = fromRate / toRate; + const newLength = Math.floor(samples.length / ratio); + const out = new Float32Array(newLength); + for (let i = 0; i < newLength; i++) { + const srcIndex = i * ratio; + const lo = Math.floor(srcIndex); + const hi = Math.min(lo + 1, samples.length - 1); + const frac = srcIndex - lo; + out[i] = samples[lo] * (1 - frac) + samples[hi] * frac; + } + return out; +} + +/** + * Build a WAV (RIFF / WAVE) blob from Float32 PCM samples in the [-1.0, 1.0] + * range. Output is 16-bit signed PCM, mono. + */ +function encodeWav(samples: Float32Array, sampleRate: number): Blob { + const dataSize = samples.length * (BITS_PER_SAMPLE / 8); + const buffer = new ArrayBuffer(RIFF_HEADER_SIZE + dataSize); + const view = new DataView(buffer); + + // RIFF header — fixed shape; getting any of these wrong makes the WAV + // unreadable by every consumer (browsers, ffmpeg, STT servers). + writeAscii(view, 0, 'RIFF'); + view.setUint32(4, 36 + dataSize, true); // file size minus the 8-byte RIFF preamble + writeAscii(view, 8, 'WAVE'); + + // fmt chunk + writeAscii(view, 12, 'fmt '); + view.setUint32(16, 16, true); // PCM fmt chunk size + view.setUint16(20, 1, true); // audio format = 1 (PCM) + view.setUint16(22, NUM_CHANNELS, true); + view.setUint32(24, sampleRate, true); + view.setUint32(28, sampleRate * NUM_CHANNELS * (BITS_PER_SAMPLE / 8), true); // byte rate + view.setUint16(32, NUM_CHANNELS * (BITS_PER_SAMPLE / 8), true); // block align + view.setUint16(34, BITS_PER_SAMPLE, true); + + // data chunk + writeAscii(view, 36, 'data'); + view.setUint32(40, dataSize, true); + + // PCM samples — clamp to [-1, 1] before scaling to int16 to avoid wrap-around + // distortion on hot signals. + let offset = RIFF_HEADER_SIZE; + for (let i = 0; i < samples.length; i++, offset += 2) { + const clamped = Math.max(-1, Math.min(1, samples[i])); + const intVal = clamped < 0 ? clamped * 0x8000 : clamped * 0x7fff; + view.setInt16(offset, intVal, true); + } + return new Blob([buffer], { type: 'audio/wav' }); +} + +function writeAscii(view: DataView, offset: number, text: string): void { + for (let i = 0; i < text.length; i++) { + view.setUint8(offset + i, text.charCodeAt(i)); + } +} diff --git a/mateclaw-ui/src/views/Settings/Stt/index.vue b/mateclaw-ui/src/views/Settings/Stt/index.vue index 7032d677..68b21113 100644 --- a/mateclaw-ui/src/views/Settings/Stt/index.vue +++ b/mateclaw-ui/src/views/Settings/Stt/index.vue @@ -28,7 +28,7 @@ @@ -57,7 +57,7 @@

- DashScope (Paraformer) + DashScope (Paraformer Realtime) {{ t('settings.sttProviderTags.reuseLlmKey') }}
{{ t('settings.hints.dashscopeSttInfo') }}
diff --git a/mateclaw-ui/vite.config.ts b/mateclaw-ui/vite.config.ts index 2abfa116..1d258a22 100644 --- a/mateclaw-ui/vite.config.ts +++ b/mateclaw-ui/vite.config.ts @@ -19,6 +19,13 @@ export default defineConfig({ '/api': { target: 'http://localhost:18088', changeOrigin: true, + // ws:true forwards WebSocket Upgrade requests through to the backend. + // Without it Vite serves the GET /api/v1/talk/ws as a regular HTTP + // proxy, the Upgrade header gets dropped, and the WS handshake + // silently fails — frontend stuck on "Connecting", backend never + // sees the connection. TalkMode (STT) lives on this WS so the + // whole feature is dead in dev mode without it. + ws: true, }, }, },