diff --git a/app.py b/app.py index a672885..bee9285 100644 --- a/app.py +++ b/app.py @@ -1,79 +1,14 @@ import gradio as gr -from faster_whisper import WhisperModel from pathlib import Path -from dotenv import load_dotenv -from typing import Optional -import os +# Подгрузка сервисов from services.llm import Llm +from services.fasterWhisper import FasterWhisper -load_dotenv() - -# Стандартный API ключ -DEFAULT_API_KEY=os.getenv('API_KEY') -# Задаем выходную директорию -OUTPUT_PATH='outputs' -# Стандартный системный промпт -DEFAULT_SYSTEM_PROMPT='''You are a diligent university student who has recorded a lecture as an audio file and later transcribed it into raw text. -Your task is to rewrite this unstructured transcript into a clear, logically organized, and detailed lecture summary (lecture notes). - -Guidelines: -1. Structure: - - Organize the text into a hierarchy of sections and subsections. - - Use headings, bullet points, or numbering where appropriate. - - Present the material in a logical flow (from introduction → main points → details → examples → conclusion). - -2. Clarity & Cohesion: - - Remove filler words, repetitions, and irrelevant fragments. - - Rewrite incomplete sentences into full, grammatically correct sentences. - - Ensure smooth transitions between topics, making the summary feel continuous and well-connected. - -3. Depth & Detail: - - Capture all important concepts, definitions, examples, and explanations from the lecture. - - Expand shorthand or fragmented thoughts into full, precise explanations. - - Where appropriate, rephrase or clarify confusing passages for better understanding. - -4. Accuracy: - - Preserve the lecturer’s original meaning, intent, and terminology. - - Avoid adding personal opinions or new information that was not in the lecture. - -5. Style: - - Write in a formal, academic tone suitable for study notes. - - Aim for readability: concise sentences, but thorough coverage of concepts. - - Use emphasis (e.g., bold or italic text) only when it improves comprehension. - -Final Output: A cohesive, detailed, and well-structured lecture summary, suitable for later studying and revision. -Use only russian language! -''' - -faster_whisper_models_list = ['tiny', 'base', 'small', 'medium', 'large-v1', 'large-v2', 'large-v3', 'large', 'distil-large-v2', 'distil-large-v3', 'distil-large-v3.5', 'large-v3-turbo', 'turbo'] -llm_models_list = ['openai/gpt-oss-120b', 'Qwen/Qwen3-235B-A22B-Thinking-2507', 'deepseek-ai/DeepSeek-R1-0528', 'meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8', 'openai/gpt-oss-20b', 'Intel/Qwen3-Coder-480B-A35B-Instruct-int4-mixed-ar', 'meta-llama/Llama-3.2-90B-Vision-Instruct', 'mistralai/Mistral-Nemo-Instruct-2407', 'Qwen/Qwen2.5-VL-32B-Instruct', 'meta-llama/Llama-3.3-70B-Instruct', 'mistralai/Devstral-Small-2505', 'mistralai/Magistral-Small-2506', 'mistralai/Mistral-Large-Instruct-2411', 'CohereForAI/aya-expanse-32b'] +# Загрузка параметров конфигурации +from config import * # Функция транскрибации -def recognize(model, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText): - model = WhisperModel(model, device='cuda', compute_type='float16') # Задаем модель - - segments, _ = model.transcribe( # Распознаем текст - audioFile, - beam_size=beamSize, - vad_filter=vadFilter, - vad_parameters={ - "min_silence_duration_ms": minSilenceDurationMs, - "speech_pad_ms": speechPadMs - }, - temperature= [temp0, temp1, temp2], - word_timestamps=wordTimestamps, - no_speech_threshold=noSpeechThreshold, - condition_on_previous_text=conditionOnPreviousText - ) - - text = '' - - for seg in segments: - text += f"[{format_timestamp(seg.start)} -> {format_timestamp(seg.end)}] {seg.text}" + '\n' - - return(text) - def generateByCondition(api_key, llm_model, system_prompt, recognized_text, llm_temperature, is_pipeline_enabled, trigger, isSaveFile): llm = Llm(api_key) @@ -98,13 +33,6 @@ def generateByCondition(api_key, llm_model, system_prompt, recognized_text, llm_ # если нет чекбокса и было событие change return gr.skip(), gr.skip() -def format_timestamp(seconds: float) -> str: - millis = int(seconds * 1000) - hours = millis // (3600 * 1000) - minutes = (millis % (3600 * 1000)) // (60 * 1000) - seconds_int = (millis % (60 * 1000)) // 1000 - millis = millis % 1000 - return f"{hours:02d}:{minutes:02d}:{seconds_int:02d},{millis:03d}" # Функция сохранения файла def saveFile(filename, text): @@ -123,7 +51,6 @@ def updateButton(isChecked): return gr.update(interactive=not isChecked, variant=variant) - def updateTextbox(isChecked): return gr.update(visible=isChecked) @@ -161,7 +88,7 @@ with gr.Blocks() as demo: with gr.Row(): # Левая колонка в акордионе with gr.Column(): - fastWhisperModel = gr.Dropdown(label='Model', choices=faster_whisper_models_list, interactive=True) + fastWhisperModel = gr.Dropdown(label='Model', choices=FAST_WHISPER_MODELS, value=FAST_WHISPER_MODELS[11], interactive=True) beamSize = gr.Number(label='beam_size', value=8, interactive=True) noSpeechThreshold = gr.Number(label='no_speech_threshold', value=0.5, interactive=True) @@ -188,7 +115,7 @@ with gr.Blocks() as demo: systemPrompt = gr.Textbox(label='', value=DEFAULT_SYSTEM_PROMPT, interactive=True) with gr.Row(): - llmModel = gr.Dropdown(label='models', choices=llm_models_list, value=llm_models_list[0], interactive=True) + llmModel = gr.Dropdown(label='models', choices=LLM_MODELS, value=LLM_MODELS[0], interactive=True) llmTemperature = gr.Number(label='Temperature', value=0.8, interactive=True ) with gr.Column(): @@ -206,7 +133,7 @@ with gr.Blocks() as demo: isPipelineEnabledCheckbox.change(updateButton, inputs=[isPipelineEnabledCheckbox], outputs=refineTextBtn) saveFileCheckbox.change(updateTextbox, inputs=saveFileCheckbox, outputs=filename) - recognizeBtn.click(recognize, outputs=[recognizedText], inputs=[fastWhisperModel, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText]) + recognizeBtn.click(FasterWhisper.recognize, outputs=[recognizedText], inputs=[fastWhisperModel, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText]) # Если пайплайн включен то тогда делаем автоматически # автоматический пайплайн @@ -223,5 +150,4 @@ with gr.Blocks() as demo: outputs=[refinedText, refinedTextMD] ) - demo.launch() diff --git a/config.py b/config.py new file mode 100644 index 0000000..f51795d --- /dev/null +++ b/config.py @@ -0,0 +1,48 @@ +import os +from dotenv import load_dotenv + +load_dotenv() + +FAST_WHISPER_MODELS = ['tiny', 'base', 'small', 'medium', 'large-v1', 'large-v2', 'large-v3', 'large', 'distil-large-v2', 'distil-large-v3', 'distil-large-v3.5', 'large-v3-turbo', 'turbo'] +LLM_MODELS = ['openai/gpt-oss-120b', 'Qwen/Qwen3-235B-A22B-Thinking-2507', 'deepseek-ai/DeepSeek-R1-0528', 'meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8', 'openai/gpt-oss-20b', 'Intel/Qwen3-Coder-480B-A35B-Instruct-int4-mixed-ar', 'meta-llama/Llama-3.2-90B-Vision-Instruct', 'mistralai/Mistral-Nemo-Instruct-2407', 'Qwen/Qwen2.5-VL-32B-Instruct', 'meta-llama/Llama-3.3-70B-Instruct', 'mistralai/Devstral-Small-2505', 'mistralai/Magistral-Small-2506', 'mistralai/Mistral-Large-Instruct-2411', 'CohereForAI/aya-expanse-32b'] + +# Стандартный API ключ +DEFAULT_API_KEY=os.getenv('API_KEY') +# Задаем выходную директорию +OUTPUT_PATH='outputs' + +# Стандартный системный промпт +OUTPUT_PATH='outputs' + +DEFAULT_SYSTEM_PROMPT='''You are a diligent university student who has recorded a lecture as an audio file and later transcribed it into raw text. +Your task is to rewrite this unstructured transcript into a clear, logically organized, and detailed lecture summary (lecture notes). + +Guidelines: +1. Structure: + - Organize the text into a hierarchy of sections and subsections. + - Use headings, bullet points, or numbering where appropriate. + - Present the material in a logical flow (from introduction → main points → details → examples → conclusion). + +2. Clarity & Cohesion: + - Remove filler words, repetitions, and irrelevant fragments. + - Rewrite incomplete sentences into full, grammatically correct sentences. + - Ensure smooth transitions between topics, making the summary feel continuous and well-connected. + +3. Depth & Detail: + - Capture all important concepts, definitions, examples, and explanations from the lecture. + - Expand shorthand or fragmented thoughts into full, precise explanations. + - Where appropriate, rephrase or clarify confusing passages for better understanding. + +4. Accuracy: + - Preserve the lecturer’s original meaning, intent, and terminology. + - Avoid adding personal opinions or new information that was not in the lecture. + +5. Style: + - Write in a formal, academic tone suitable for study notes. + - Aim for readability: concise sentences, but thorough coverage of concepts. + - Use emphasis (e.g., bold or italic text) only when it improves comprehension. + +Final Output: A cohesive, detailed, and well-structured lecture summary, suitable for later studying and revision. +Use only russian language! +''' + diff --git a/services/__init__.py b/services/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/services/fasterWhisper.py b/services/fasterWhisper.py new file mode 100644 index 0000000..9072b53 --- /dev/null +++ b/services/fasterWhisper.py @@ -0,0 +1,39 @@ +from faster_whisper import WhisperModel + +class FasterWhisper: + def __init__(self): + pass + + def recognize(self, model, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText): + model = WhisperModel(model, device='cuda', compute_type='float16') # Задаем модель + + segments, _ = model.transcribe( # Распознаем текст + audioFile, + beam_size=beamSize, + vad_filter=vadFilter, + vad_parameters={ + "min_silence_duration_ms": minSilenceDurationMs, + "speech_pad_ms": speechPadMs + }, + temperature= [temp0, temp1, temp2], + word_timestamps=wordTimestamps, + no_speech_threshold=noSpeechThreshold, + condition_on_previous_text=conditionOnPreviousText + ) + + text = '' + + for seg in segments: + text += f"[{self.format_timestamp(seg.start)} -> {self.format_timestamp(seg.end)}] {seg.text}" + '\n' + + return(text) + + def format_timestamp(self, seconds: float) -> str: + millis = int(seconds * 1000) + hours = millis // (3600 * 1000) + minutes = (millis % (3600 * 1000)) // (60 * 1000) + seconds_int = (millis % (60 * 1000)) // 1000 + millis = millis % 1000 + return f"{hours:02d}:{minutes:02d}:{seconds_int:02d},{millis:03d}" + +