add device and compute_type in gradio
This commit is contained in:
6
app.py
6
app.py
@@ -128,6 +128,8 @@ with gr.Blocks() as demo:
|
||||
with gr.Row():
|
||||
# Левая колонка в акордионе
|
||||
with gr.Column():
|
||||
device = gr.Dropdown(label='Device', choices=["cpu", "cuda"], value="cuda", interactive=True)
|
||||
compute_type = gr.Dropdown(label='compute_type', choices=["auto", "int8", "float16", "float32"], value="auto", interactive=True)
|
||||
fastWhisperModel = gr.Dropdown(label='Model', choices=FAST_WHISPER_MODELS, value=FAST_WHISPER_MODELS[11], interactive=True)
|
||||
|
||||
beamSize = gr.Number(label='beam_size', value=8, interactive=True)
|
||||
@@ -136,6 +138,8 @@ with gr.Blocks() as demo:
|
||||
wordTimestamps = gr.Checkbox(label='word_timestamps', value=True, interactive=True)
|
||||
conditionOnPreviousText = gr.Checkbox(label='condition_on_previous_text', value=False, interactive=True)
|
||||
|
||||
|
||||
|
||||
# Правая колонка в акордионе
|
||||
with gr.Column():
|
||||
with gr.Accordion(label='Vad parameters'):
|
||||
@@ -171,7 +175,7 @@ with gr.Blocks() as demo:
|
||||
saveFileCheckbox.change(updateTextbox, inputs=saveFileCheckbox, outputs=filename)
|
||||
saveFileCheckbox.change(updateTextbox, inputs=saveFileCheckbox, outputs=filenamePdf)
|
||||
|
||||
recognizeBtn.click(FasterWhisper().recognize, outputs=[recognizedText], inputs=[fastWhisperModel, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText])
|
||||
recognizeBtn.click(FasterWhisper().recognize, outputs=[recognizedText], inputs=[fastWhisperModel, device, compute_type, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText])
|
||||
|
||||
# Если пайплайн включен то тогда делаем автоматически
|
||||
# автоматический пайплайн
|
||||
|
||||
@@ -6,6 +6,7 @@ load_dotenv()
|
||||
FAST_WHISPER_MODELS = ['tiny', 'base', 'small', 'medium', 'large-v1', 'large-v2', 'large-v3', 'large', 'distil-large-v2', 'distil-large-v3', 'distil-large-v3.5', 'large-v3-turbo', 'turbo']
|
||||
LLM_MODELS = ['openai/gpt-oss-120b', 'Qwen/Qwen3-235B-A22B-Thinking-2507', 'deepseek-ai/DeepSeek-R1-0528', 'meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8', 'openai/gpt-oss-20b', 'Intel/Qwen3-Coder-480B-A35B-Instruct-int4-mixed-ar', 'meta-llama/Llama-3.2-90B-Vision-Instruct', 'mistralai/Mistral-Nemo-Instruct-2407', 'Qwen/Qwen2.5-VL-32B-Instruct', 'meta-llama/Llama-3.3-70B-Instruct', 'mistralai/Devstral-Small-2505', 'mistralai/Magistral-Small-2506', 'mistralai/Mistral-Large-Instruct-2411', 'CohereForAI/aya-expanse-32b']
|
||||
|
||||
|
||||
# Стандартный API ключ
|
||||
DEFAULT_API_KEY=os.getenv('API_KEY')
|
||||
# Задаем выходную директорию
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
from faster_whisper import WhisperModel
|
||||
|
||||
class FasterWhisper:
|
||||
def recognize(self, model, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText):
|
||||
model = WhisperModel(model, device='cuda', compute_type='auto') # Задаем модель
|
||||
def recognize(self, model, device, compute_type, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText):
|
||||
model = WhisperModel(model, device=device, compute_type=compute_type) # Задаем модель
|
||||
|
||||
segments, _ = model.transcribe( # Распознаем текст
|
||||
audioFile,
|
||||
|
||||
Reference in New Issue
Block a user