Support using GenAI for audio transcription (#24396)
CI / AMD64 Build (push) Canceled after 0s
CI / AMD64 Smoke Test (push) Canceled after 0s
CI / ARM Build (push) Canceled after 0s
CI / Jetson Jetpack 6 (push) Canceled after 0s
CI / AMD64 Extra Build (push) Canceled after 0s
CI / ARM Extra Build (push) Canceled after 0s
CI / Synaptics Build (push) Canceled after 0s
CI / Assemble and push default build (push) Canceled after 0s

* Add support for running transcription with GenAI

* Improve audio joining

* Fix GenAI model capability reporting

* Support language correctly

* Migrate existing users to keep english selected

* Fix models

* Fix tests

* Fix accepted null model

* Handle slwo providers
This commit is contained in:
Nicolas Mowen
2026-09-17 16:34:47 -05:00
committed by GitHub
parent eccd10cd94
commit 334073967b
35 changed files with 2563 additions and 278 deletions
+6 -2
View File
@@ -348,7 +348,7 @@
},
"roles": {
"label": "Roles",
"description": "GenAI roles (chat, descriptions, embeddings); one provider per role."
"description": "GenAI roles (chat, descriptions, embeddings, transcribe); one provider per role. Only chat, descriptions, and embeddings are granted by default; transcribe must be listed explicitly."
},
"provider_options": {
"label": "Provider options",
@@ -1131,7 +1131,11 @@
},
"language": {
"label": "Transcription language",
"description": "Language code used for transcription/translation (for example 'en' for English). See https://whisper-api.com/docs/languages/ for supported language codes."
"description": "Language code used for transcription/translation (for example 'en' for English), or 'auto' to let the model detect it. See https://whisper-api.com/docs/languages/ for supported language codes."
},
"model": {
"label": "Audio transcription model or GenAI provider name",
"description": "The transcription backend: 'whisper' for Frigate's built-in local models, or the name of a GenAI provider with the transcribe role."
},
"device": {
"label": "Transcription device",
+12 -2
View File
@@ -1693,7 +1693,8 @@
"options": {
"embeddings": "Embedding",
"descriptions": "Descriptions",
"chat": "Chat"
"chat": "Chat",
"transcribe": "Transcription"
}
},
"semanticSearchModel": {
@@ -1742,6 +1743,14 @@
"admin": "Admin",
"viewer": "Viewer",
"none": "None (deny access)"
},
"audioTranscriptionModel": {
"placeholder": "Select model…",
"builtIn": "Built-in Models",
"genaiProviders": "GenAI Providers"
},
"audioTranscriptionModelSize": {
"notApplicable": "Not applicable for GenAI providers"
}
},
"globalConfig": {
@@ -1953,7 +1962,8 @@
"noAudioRole": "No streams have the audio role defined. You must enable the audio role for audio detection to function."
},
"audioTranscription": {
"audioDetectionDisabled": "Audio detection is not enabled for this camera. Audio transcription requires audio detection to be active."
"audioDetectionDisabled": "Audio detection is not enabled for this camera. Audio transcription requires audio detection to be active.",
"genaiProviderSelected": "A GenAI provider is selected, so the device and model size settings are ignored."
},
"detect": {
"fpsGreaterThanFive": "Setting the detect FPS higher than 5 is not recommended. Higher values may cause performance issues and will not provide any benefit.",