From 85b61c650c87946e45c297194e89c76dc4569a28 Mon Sep 17 00:00:00 2001
From: alexliluz <49665315+alexliluz@users.noreply.github.com>
Date: Wed, 19 Aug 2026 20:37:29 +0800
Subject: [PATCH] feat(audio): cap voice recording duration
Add an optional admin-configured duration limit for voice input and automatically submit the captured recording when the limit is reached.
Refs #28715
Assisted-by: ChatGPT 5.2
---
backend/open_webui/config.py | 5 +++++
backend/open_webui/main.py | 2 ++
backend/open_webui/routers/audio.py | 2 ++
.../components/admin/Settings/Audio.svelte | 21 +++++++++++++++++++
src/lib/components/chat/MessageInput.svelte | 1 +
.../chat/MessageInput/VoiceRecording.svelte | 4 ++++
6 files changed, 35 insertions(+)
diff --git a/backend/open_webui/config.py b/backend/open_webui/config.py
index 75b4b6d284..14aa3ad289 100644
--- a/backend/open_webui/config.py
+++ b/backend/open_webui/config.py
@@ -1552,6 +1552,10 @@ AUDIO_STT_ENGINE = os.getenv('AUDIO_STT_ENGINE', '')
AUDIO_STT_MODEL = os.getenv('AUDIO_STT_MODEL', '')
+AUDIO_STT_MAX_RECORDING_DURATION = (
+ int(os.getenv('AUDIO_STT_MAX_RECORDING_DURATION')) if os.getenv('AUDIO_STT_MAX_RECORDING_DURATION') else None
+)
+
AUDIO_STT_SUPPORTED_CONTENT_TYPES = [
content_type.strip()
for content_type in os.getenv('AUDIO_STT_SUPPORTED_CONTENT_TYPES', '').split(',')
@@ -3045,6 +3049,7 @@ DEFAULT_CONFIG = {
'audio.stt.openai.api_request_format': AUDIO_STT_OPENAI_API_REQUEST_FORMAT,
'audio.stt.engine': AUDIO_STT_ENGINE,
'audio.stt.model': AUDIO_STT_MODEL,
+ 'audio.stt.max_recording_duration': AUDIO_STT_MAX_RECORDING_DURATION,
'audio.stt.supported_content_types': AUDIO_STT_SUPPORTED_CONTENT_TYPES,
'audio.stt.allowed_extensions': AUDIO_STT_ALLOWED_EXTENSIONS,
'audio.stt.azure.api_key': AUDIO_STT_AZURE_API_KEY,
diff --git a/backend/open_webui/main.py b/backend/open_webui/main.py
index 6cbf0155cb..6219be7929 100644
--- a/backend/open_webui/main.py
+++ b/backend/open_webui/main.py
@@ -2261,6 +2261,7 @@ async def get_app_config(request: Request):
'audio.tts.voice',
'audio.tts.split_on',
'audio.stt.engine',
+ 'audio.stt.max_recording_duration',
'rag.file.max_size',
'rag.file.max_count',
'file.image_compression_width',
@@ -2363,6 +2364,7 @@ async def get_app_config(request: Request):
},
'stt': {
'engine': config.get('audio.stt.engine'),
+ 'max_recording_duration': config.get('audio.stt.max_recording_duration'),
},
},
'file': {
diff --git a/backend/open_webui/routers/audio.py b/backend/open_webui/routers/audio.py
index 9fa5d89304..e4db17e25d 100644
--- a/backend/open_webui/routers/audio.py
+++ b/backend/open_webui/routers/audio.py
@@ -97,6 +97,7 @@ STT_CONFIG_KEYS = {
'OPENAI_API_REQUEST_FORMAT': 'audio.stt.openai.api_request_format',
'ENGINE': 'audio.stt.engine',
'MODEL': 'audio.stt.model',
+ 'MAX_RECORDING_DURATION': 'audio.stt.max_recording_duration',
'SUPPORTED_CONTENT_TYPES': 'audio.stt.supported_content_types',
'ALLOWED_EXTENSIONS': 'audio.stt.allowed_extensions',
'WHISPER_MODEL': 'audio.stt.whisper_model',
@@ -256,6 +257,7 @@ class STTConfigForm(BaseModel):
OPENAI_API_REQUEST_FORMAT: str = 'multipart'
ENGINE: str
MODEL: str
+ MAX_RECORDING_DURATION: Optional[int] = None
SUPPORTED_CONTENT_TYPES: list[str] = []
ALLOWED_EXTENSIONS: list[str] = []
WHISPER_MODEL: str
diff --git a/src/lib/components/admin/Settings/Audio.svelte b/src/lib/components/admin/Settings/Audio.svelte
index e09e4aee3d..963ae39267 100644
--- a/src/lib/components/admin/Settings/Audio.svelte
+++ b/src/lib/components/admin/Settings/Audio.svelte
@@ -51,6 +51,7 @@
let STT_OPENAI_API_REQUEST_FORMAT = 'multipart';
let STT_ENGINE = '';
let STT_MODEL = '';
+ let STT_MAX_RECORDING_DURATION: number | '' = '';
let STT_SUPPORTED_CONTENT_TYPES = '';
let STT_WHISPER_MODEL = '';
let STT_AZURE_API_KEY = '';
@@ -165,6 +166,9 @@
OPENAI_API_REQUEST_FORMAT: STT_OPENAI_API_REQUEST_FORMAT,
ENGINE: STT_ENGINE,
MODEL: STT_MODEL,
+ MAX_RECORDING_DURATION: STT_MAX_RECORDING_DURATION
+ ? Number(STT_MAX_RECORDING_DURATION)
+ : null,
SUPPORTED_CONTENT_TYPES: STT_SUPPORTED_CONTENT_TYPES.split(','),
WHISPER_MODEL: STT_WHISPER_MODEL,
DEEPGRAM_API_KEY: STT_DEEPGRAM_API_KEY,
@@ -219,6 +223,7 @@
STT_ENGINE = res.stt.ENGINE;
STT_MODEL = res.stt.MODEL;
+ STT_MAX_RECORDING_DURATION = res.stt.MAX_RECORDING_DURATION ?? '';
STT_SUPPORTED_CONTENT_TYPES = (res?.stt?.SUPPORTED_CONTENT_TYPES ?? []).join(',');
STT_WHISPER_MODEL = res.stt.WHISPER_MODEL;
STT_AZURE_API_KEY = res.stt.AZURE_API_KEY;
@@ -262,6 +267,22 @@
+
+
+
+
{#if STT_ENGINE !== 'web'}
{
recording = false;
diff --git a/src/lib/components/chat/MessageInput/VoiceRecording.svelte b/src/lib/components/chat/MessageInput/VoiceRecording.svelte
index 88048ec1a2..de0b6c69a2 100644
--- a/src/lib/components/chat/MessageInput/VoiceRecording.svelte
+++ b/src/lib/components/chat/MessageInput/VoiceRecording.svelte
@@ -20,6 +20,7 @@
export let echoCancellation = true;
export let noiseSuppression = true;
export let autoGainControl = true;
+ export let maxDurationSeconds = 0;
export let className = ' p-2.5 w-full max-w-full';
@@ -37,6 +38,9 @@
const startDurationCounter = () => {
durationCounter = setInterval(() => {
durationSeconds++;
+ if (maxDurationSeconds > 0 && durationSeconds >= maxDurationSeconds && recording && !loading) {
+ void confirmRecording();
+ }
}, 1000);
};