From 85b61c650c87946e45c297194e89c76dc4569a28 Mon Sep 17 00:00:00 2001 From: alexliluz <49665315+alexliluz@users.noreply.github.com> Date: Wed, 19 Aug 2026 20:37:29 +0800 Subject: [PATCH] feat(audio): cap voice recording duration Add an optional admin-configured duration limit for voice input and automatically submit the captured recording when the limit is reached. Refs #28715 Assisted-by: ChatGPT 5.2 --- backend/open_webui/config.py | 5 +++++ backend/open_webui/main.py | 2 ++ backend/open_webui/routers/audio.py | 2 ++ .../components/admin/Settings/Audio.svelte | 21 +++++++++++++++++++ src/lib/components/chat/MessageInput.svelte | 1 + .../chat/MessageInput/VoiceRecording.svelte | 4 ++++ 6 files changed, 35 insertions(+) diff --git a/backend/open_webui/config.py b/backend/open_webui/config.py index 75b4b6d284..14aa3ad289 100644 --- a/backend/open_webui/config.py +++ b/backend/open_webui/config.py @@ -1552,6 +1552,10 @@ AUDIO_STT_ENGINE = os.getenv('AUDIO_STT_ENGINE', '') AUDIO_STT_MODEL = os.getenv('AUDIO_STT_MODEL', '') +AUDIO_STT_MAX_RECORDING_DURATION = ( + int(os.getenv('AUDIO_STT_MAX_RECORDING_DURATION')) if os.getenv('AUDIO_STT_MAX_RECORDING_DURATION') else None +) + AUDIO_STT_SUPPORTED_CONTENT_TYPES = [ content_type.strip() for content_type in os.getenv('AUDIO_STT_SUPPORTED_CONTENT_TYPES', '').split(',') @@ -3045,6 +3049,7 @@ DEFAULT_CONFIG = { 'audio.stt.openai.api_request_format': AUDIO_STT_OPENAI_API_REQUEST_FORMAT, 'audio.stt.engine': AUDIO_STT_ENGINE, 'audio.stt.model': AUDIO_STT_MODEL, + 'audio.stt.max_recording_duration': AUDIO_STT_MAX_RECORDING_DURATION, 'audio.stt.supported_content_types': AUDIO_STT_SUPPORTED_CONTENT_TYPES, 'audio.stt.allowed_extensions': AUDIO_STT_ALLOWED_EXTENSIONS, 'audio.stt.azure.api_key': AUDIO_STT_AZURE_API_KEY, diff --git a/backend/open_webui/main.py b/backend/open_webui/main.py index 6cbf0155cb..6219be7929 100644 --- a/backend/open_webui/main.py +++ b/backend/open_webui/main.py @@ -2261,6 +2261,7 @@ async def get_app_config(request: Request): 'audio.tts.voice', 'audio.tts.split_on', 'audio.stt.engine', + 'audio.stt.max_recording_duration', 'rag.file.max_size', 'rag.file.max_count', 'file.image_compression_width', @@ -2363,6 +2364,7 @@ async def get_app_config(request: Request): }, 'stt': { 'engine': config.get('audio.stt.engine'), + 'max_recording_duration': config.get('audio.stt.max_recording_duration'), }, }, 'file': { diff --git a/backend/open_webui/routers/audio.py b/backend/open_webui/routers/audio.py index 9fa5d89304..e4db17e25d 100644 --- a/backend/open_webui/routers/audio.py +++ b/backend/open_webui/routers/audio.py @@ -97,6 +97,7 @@ STT_CONFIG_KEYS = { 'OPENAI_API_REQUEST_FORMAT': 'audio.stt.openai.api_request_format', 'ENGINE': 'audio.stt.engine', 'MODEL': 'audio.stt.model', + 'MAX_RECORDING_DURATION': 'audio.stt.max_recording_duration', 'SUPPORTED_CONTENT_TYPES': 'audio.stt.supported_content_types', 'ALLOWED_EXTENSIONS': 'audio.stt.allowed_extensions', 'WHISPER_MODEL': 'audio.stt.whisper_model', @@ -256,6 +257,7 @@ class STTConfigForm(BaseModel): OPENAI_API_REQUEST_FORMAT: str = 'multipart' ENGINE: str MODEL: str + MAX_RECORDING_DURATION: Optional[int] = None SUPPORTED_CONTENT_TYPES: list[str] = [] ALLOWED_EXTENSIONS: list[str] = [] WHISPER_MODEL: str diff --git a/src/lib/components/admin/Settings/Audio.svelte b/src/lib/components/admin/Settings/Audio.svelte index e09e4aee3d..963ae39267 100644 --- a/src/lib/components/admin/Settings/Audio.svelte +++ b/src/lib/components/admin/Settings/Audio.svelte @@ -51,6 +51,7 @@ let STT_OPENAI_API_REQUEST_FORMAT = 'multipart'; let STT_ENGINE = ''; let STT_MODEL = ''; + let STT_MAX_RECORDING_DURATION: number | '' = ''; let STT_SUPPORTED_CONTENT_TYPES = ''; let STT_WHISPER_MODEL = ''; let STT_AZURE_API_KEY = ''; @@ -165,6 +166,9 @@ OPENAI_API_REQUEST_FORMAT: STT_OPENAI_API_REQUEST_FORMAT, ENGINE: STT_ENGINE, MODEL: STT_MODEL, + MAX_RECORDING_DURATION: STT_MAX_RECORDING_DURATION + ? Number(STT_MAX_RECORDING_DURATION) + : null, SUPPORTED_CONTENT_TYPES: STT_SUPPORTED_CONTENT_TYPES.split(','), WHISPER_MODEL: STT_WHISPER_MODEL, DEEPGRAM_API_KEY: STT_DEEPGRAM_API_KEY, @@ -219,6 +223,7 @@ STT_ENGINE = res.stt.ENGINE; STT_MODEL = res.stt.MODEL; + STT_MAX_RECORDING_DURATION = res.stt.MAX_RECORDING_DURATION ?? ''; STT_SUPPORTED_CONTENT_TYPES = (res?.stt?.SUPPORTED_CONTENT_TYPES ?? []).join(','); STT_WHISPER_MODEL = res.stt.WHISPER_MODEL; STT_AZURE_API_KEY = res.stt.AZURE_API_KEY; @@ -262,6 +267,22 @@ + + + + {#if STT_ENGINE !== 'web'} { recording = false; diff --git a/src/lib/components/chat/MessageInput/VoiceRecording.svelte b/src/lib/components/chat/MessageInput/VoiceRecording.svelte index 88048ec1a2..de0b6c69a2 100644 --- a/src/lib/components/chat/MessageInput/VoiceRecording.svelte +++ b/src/lib/components/chat/MessageInput/VoiceRecording.svelte @@ -20,6 +20,7 @@ export let echoCancellation = true; export let noiseSuppression = true; export let autoGainControl = true; + export let maxDurationSeconds = 0; export let className = ' p-2.5 w-full max-w-full'; @@ -37,6 +38,9 @@ const startDurationCounter = () => { durationCounter = setInterval(() => { durationSeconds++; + if (maxDurationSeconds > 0 && durationSeconds >= maxDurationSeconds && recording && !loading) { + void confirmRecording(); + } }, 1000); };