diff --git a/app.py b/app.py index 906d2a4..d0b344d 100644 --- a/app.py +++ b/app.py @@ -128,6 +128,8 @@ with gr.Blocks() as demo: with gr.Row(): # Левая колонка в акордионе with gr.Column(): + device = gr.Dropdown(label='Device', choices=["cpu", "cuda"], value="cuda", interactive=True) + compute_type = gr.Dropdown(label='compute_type', choices=["auto", "int8", "float16", "float32"], value="auto", interactive=True) fastWhisperModel = gr.Dropdown(label='Model', choices=FAST_WHISPER_MODELS, value=FAST_WHISPER_MODELS[11], interactive=True) beamSize = gr.Number(label='beam_size', value=8, interactive=True) @@ -136,6 +138,8 @@ with gr.Blocks() as demo: wordTimestamps = gr.Checkbox(label='word_timestamps', value=True, interactive=True) conditionOnPreviousText = gr.Checkbox(label='condition_on_previous_text', value=False, interactive=True) + + # Правая колонка в акордионе with gr.Column(): with gr.Accordion(label='Vad parameters'): @@ -171,7 +175,7 @@ with gr.Blocks() as demo: saveFileCheckbox.change(updateTextbox, inputs=saveFileCheckbox, outputs=filename) saveFileCheckbox.change(updateTextbox, inputs=saveFileCheckbox, outputs=filenamePdf) - recognizeBtn.click(FasterWhisper().recognize, outputs=[recognizedText], inputs=[fastWhisperModel, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText]) + recognizeBtn.click(FasterWhisper().recognize, outputs=[recognizedText], inputs=[fastWhisperModel, device, compute_type, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText]) # Если пайплайн включен то тогда делаем автоматически # автоматический пайплайн diff --git a/config.py b/config.py index cf9cac1..05c4978 100644 --- a/config.py +++ b/config.py @@ -6,6 +6,7 @@ load_dotenv() FAST_WHISPER_MODELS = ['tiny', 'base', 'small', 'medium', 'large-v1', 'large-v2', 'large-v3', 'large', 'distil-large-v2', 'distil-large-v3', 'distil-large-v3.5', 'large-v3-turbo', 'turbo'] LLM_MODELS = ['openai/gpt-oss-120b', 'Qwen/Qwen3-235B-A22B-Thinking-2507', 'deepseek-ai/DeepSeek-R1-0528', 'meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8', 'openai/gpt-oss-20b', 'Intel/Qwen3-Coder-480B-A35B-Instruct-int4-mixed-ar', 'meta-llama/Llama-3.2-90B-Vision-Instruct', 'mistralai/Mistral-Nemo-Instruct-2407', 'Qwen/Qwen2.5-VL-32B-Instruct', 'meta-llama/Llama-3.3-70B-Instruct', 'mistralai/Devstral-Small-2505', 'mistralai/Magistral-Small-2506', 'mistralai/Mistral-Large-Instruct-2411', 'CohereForAI/aya-expanse-32b'] + # Стандартный API ключ DEFAULT_API_KEY=os.getenv('API_KEY') # Задаем выходную директорию diff --git a/services/fasterWhisper.py b/services/fasterWhisper.py index 2c7c24b..882468d 100644 --- a/services/fasterWhisper.py +++ b/services/fasterWhisper.py @@ -1,8 +1,8 @@ from faster_whisper import WhisperModel class FasterWhisper: - def recognize(self, model, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText): - model = WhisperModel(model, device='cuda', compute_type='auto') # Задаем модель + def recognize(self, model, device, compute_type, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText): + model = WhisperModel(model, device=device, compute_type=compute_type) # Задаем модель segments, _ = model.transcribe( # Распознаем текст audioFile,