change structure of the project
Changes: Move some modules for it's own file and move config to the `config.py`
This commit is contained in:
88
app.py
88
app.py
@@ -1,79 +1,14 @@
|
|||||||
import gradio as gr
|
import gradio as gr
|
||||||
from faster_whisper import WhisperModel
|
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from dotenv import load_dotenv
|
|
||||||
from typing import Optional
|
|
||||||
import os
|
|
||||||
|
|
||||||
|
# Подгрузка сервисов
|
||||||
from services.llm import Llm
|
from services.llm import Llm
|
||||||
|
from services.fasterWhisper import FasterWhisper
|
||||||
|
|
||||||
load_dotenv()
|
# Загрузка параметров конфигурации
|
||||||
|
from config import *
|
||||||
# Стандартный API ключ
|
|
||||||
DEFAULT_API_KEY=os.getenv('API_KEY')
|
|
||||||
# Задаем выходную директорию
|
|
||||||
OUTPUT_PATH='outputs'
|
|
||||||
# Стандартный системный промпт
|
|
||||||
DEFAULT_SYSTEM_PROMPT='''You are a diligent university student who has recorded a lecture as an audio file and later transcribed it into raw text.
|
|
||||||
Your task is to rewrite this unstructured transcript into a clear, logically organized, and detailed lecture summary (lecture notes).
|
|
||||||
|
|
||||||
Guidelines:
|
|
||||||
1. Structure:
|
|
||||||
- Organize the text into a hierarchy of sections and subsections.
|
|
||||||
- Use headings, bullet points, or numbering where appropriate.
|
|
||||||
- Present the material in a logical flow (from introduction → main points → details → examples → conclusion).
|
|
||||||
|
|
||||||
2. Clarity & Cohesion:
|
|
||||||
- Remove filler words, repetitions, and irrelevant fragments.
|
|
||||||
- Rewrite incomplete sentences into full, grammatically correct sentences.
|
|
||||||
- Ensure smooth transitions between topics, making the summary feel continuous and well-connected.
|
|
||||||
|
|
||||||
3. Depth & Detail:
|
|
||||||
- Capture all important concepts, definitions, examples, and explanations from the lecture.
|
|
||||||
- Expand shorthand or fragmented thoughts into full, precise explanations.
|
|
||||||
- Where appropriate, rephrase or clarify confusing passages for better understanding.
|
|
||||||
|
|
||||||
4. Accuracy:
|
|
||||||
- Preserve the lecturer’s original meaning, intent, and terminology.
|
|
||||||
- Avoid adding personal opinions or new information that was not in the lecture.
|
|
||||||
|
|
||||||
5. Style:
|
|
||||||
- Write in a formal, academic tone suitable for study notes.
|
|
||||||
- Aim for readability: concise sentences, but thorough coverage of concepts.
|
|
||||||
- Use emphasis (e.g., bold or italic text) only when it improves comprehension.
|
|
||||||
|
|
||||||
Final Output: A cohesive, detailed, and well-structured lecture summary, suitable for later studying and revision.
|
|
||||||
Use only russian language!
|
|
||||||
'''
|
|
||||||
|
|
||||||
faster_whisper_models_list = ['tiny', 'base', 'small', 'medium', 'large-v1', 'large-v2', 'large-v3', 'large', 'distil-large-v2', 'distil-large-v3', 'distil-large-v3.5', 'large-v3-turbo', 'turbo']
|
|
||||||
llm_models_list = ['openai/gpt-oss-120b', 'Qwen/Qwen3-235B-A22B-Thinking-2507', 'deepseek-ai/DeepSeek-R1-0528', 'meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8', 'openai/gpt-oss-20b', 'Intel/Qwen3-Coder-480B-A35B-Instruct-int4-mixed-ar', 'meta-llama/Llama-3.2-90B-Vision-Instruct', 'mistralai/Mistral-Nemo-Instruct-2407', 'Qwen/Qwen2.5-VL-32B-Instruct', 'meta-llama/Llama-3.3-70B-Instruct', 'mistralai/Devstral-Small-2505', 'mistralai/Magistral-Small-2506', 'mistralai/Mistral-Large-Instruct-2411', 'CohereForAI/aya-expanse-32b']
|
|
||||||
|
|
||||||
# Функция транскрибации
|
# Функция транскрибации
|
||||||
def recognize(model, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText):
|
|
||||||
model = WhisperModel(model, device='cuda', compute_type='float16') # Задаем модель
|
|
||||||
|
|
||||||
segments, _ = model.transcribe( # Распознаем текст
|
|
||||||
audioFile,
|
|
||||||
beam_size=beamSize,
|
|
||||||
vad_filter=vadFilter,
|
|
||||||
vad_parameters={
|
|
||||||
"min_silence_duration_ms": minSilenceDurationMs,
|
|
||||||
"speech_pad_ms": speechPadMs
|
|
||||||
},
|
|
||||||
temperature= [temp0, temp1, temp2],
|
|
||||||
word_timestamps=wordTimestamps,
|
|
||||||
no_speech_threshold=noSpeechThreshold,
|
|
||||||
condition_on_previous_text=conditionOnPreviousText
|
|
||||||
)
|
|
||||||
|
|
||||||
text = ''
|
|
||||||
|
|
||||||
for seg in segments:
|
|
||||||
text += f"[{format_timestamp(seg.start)} -> {format_timestamp(seg.end)}] {seg.text}" + '\n'
|
|
||||||
|
|
||||||
return(text)
|
|
||||||
|
|
||||||
def generateByCondition(api_key, llm_model, system_prompt, recognized_text, llm_temperature, is_pipeline_enabled, trigger, isSaveFile):
|
def generateByCondition(api_key, llm_model, system_prompt, recognized_text, llm_temperature, is_pipeline_enabled, trigger, isSaveFile):
|
||||||
llm = Llm(api_key)
|
llm = Llm(api_key)
|
||||||
|
|
||||||
@@ -98,13 +33,6 @@ def generateByCondition(api_key, llm_model, system_prompt, recognized_text, llm_
|
|||||||
# если нет чекбокса и было событие change
|
# если нет чекбокса и было событие change
|
||||||
return gr.skip(), gr.skip()
|
return gr.skip(), gr.skip()
|
||||||
|
|
||||||
def format_timestamp(seconds: float) -> str:
|
|
||||||
millis = int(seconds * 1000)
|
|
||||||
hours = millis // (3600 * 1000)
|
|
||||||
minutes = (millis % (3600 * 1000)) // (60 * 1000)
|
|
||||||
seconds_int = (millis % (60 * 1000)) // 1000
|
|
||||||
millis = millis % 1000
|
|
||||||
return f"{hours:02d}:{minutes:02d}:{seconds_int:02d},{millis:03d}"
|
|
||||||
|
|
||||||
# Функция сохранения файла
|
# Функция сохранения файла
|
||||||
def saveFile(filename, text):
|
def saveFile(filename, text):
|
||||||
@@ -123,7 +51,6 @@ def updateButton(isChecked):
|
|||||||
return gr.update(interactive=not isChecked, variant=variant)
|
return gr.update(interactive=not isChecked, variant=variant)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
def updateTextbox(isChecked):
|
def updateTextbox(isChecked):
|
||||||
return gr.update(visible=isChecked)
|
return gr.update(visible=isChecked)
|
||||||
|
|
||||||
@@ -161,7 +88,7 @@ with gr.Blocks() as demo:
|
|||||||
with gr.Row():
|
with gr.Row():
|
||||||
# Левая колонка в акордионе
|
# Левая колонка в акордионе
|
||||||
with gr.Column():
|
with gr.Column():
|
||||||
fastWhisperModel = gr.Dropdown(label='Model', choices=faster_whisper_models_list, interactive=True)
|
fastWhisperModel = gr.Dropdown(label='Model', choices=FAST_WHISPER_MODELS, value=FAST_WHISPER_MODELS[11], interactive=True)
|
||||||
|
|
||||||
beamSize = gr.Number(label='beam_size', value=8, interactive=True)
|
beamSize = gr.Number(label='beam_size', value=8, interactive=True)
|
||||||
noSpeechThreshold = gr.Number(label='no_speech_threshold', value=0.5, interactive=True)
|
noSpeechThreshold = gr.Number(label='no_speech_threshold', value=0.5, interactive=True)
|
||||||
@@ -188,7 +115,7 @@ with gr.Blocks() as demo:
|
|||||||
systemPrompt = gr.Textbox(label='', value=DEFAULT_SYSTEM_PROMPT, interactive=True)
|
systemPrompt = gr.Textbox(label='', value=DEFAULT_SYSTEM_PROMPT, interactive=True)
|
||||||
|
|
||||||
with gr.Row():
|
with gr.Row():
|
||||||
llmModel = gr.Dropdown(label='models', choices=llm_models_list, value=llm_models_list[0], interactive=True)
|
llmModel = gr.Dropdown(label='models', choices=LLM_MODELS, value=LLM_MODELS[0], interactive=True)
|
||||||
llmTemperature = gr.Number(label='Temperature', value=0.8, interactive=True )
|
llmTemperature = gr.Number(label='Temperature', value=0.8, interactive=True )
|
||||||
|
|
||||||
with gr.Column():
|
with gr.Column():
|
||||||
@@ -206,7 +133,7 @@ with gr.Blocks() as demo:
|
|||||||
isPipelineEnabledCheckbox.change(updateButton, inputs=[isPipelineEnabledCheckbox], outputs=refineTextBtn)
|
isPipelineEnabledCheckbox.change(updateButton, inputs=[isPipelineEnabledCheckbox], outputs=refineTextBtn)
|
||||||
saveFileCheckbox.change(updateTextbox, inputs=saveFileCheckbox, outputs=filename)
|
saveFileCheckbox.change(updateTextbox, inputs=saveFileCheckbox, outputs=filename)
|
||||||
|
|
||||||
recognizeBtn.click(recognize, outputs=[recognizedText], inputs=[fastWhisperModel, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText])
|
recognizeBtn.click(FasterWhisper.recognize, outputs=[recognizedText], inputs=[fastWhisperModel, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText])
|
||||||
|
|
||||||
# Если пайплайн включен то тогда делаем автоматически
|
# Если пайплайн включен то тогда делаем автоматически
|
||||||
# автоматический пайплайн
|
# автоматический пайплайн
|
||||||
@@ -223,5 +150,4 @@ with gr.Blocks() as demo:
|
|||||||
outputs=[refinedText, refinedTextMD]
|
outputs=[refinedText, refinedTextMD]
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
demo.launch()
|
demo.launch()
|
||||||
|
|||||||
48
config.py
Normal file
48
config.py
Normal file
@@ -0,0 +1,48 @@
|
|||||||
|
import os
|
||||||
|
from dotenv import load_dotenv
|
||||||
|
|
||||||
|
load_dotenv()
|
||||||
|
|
||||||
|
FAST_WHISPER_MODELS = ['tiny', 'base', 'small', 'medium', 'large-v1', 'large-v2', 'large-v3', 'large', 'distil-large-v2', 'distil-large-v3', 'distil-large-v3.5', 'large-v3-turbo', 'turbo']
|
||||||
|
LLM_MODELS = ['openai/gpt-oss-120b', 'Qwen/Qwen3-235B-A22B-Thinking-2507', 'deepseek-ai/DeepSeek-R1-0528', 'meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8', 'openai/gpt-oss-20b', 'Intel/Qwen3-Coder-480B-A35B-Instruct-int4-mixed-ar', 'meta-llama/Llama-3.2-90B-Vision-Instruct', 'mistralai/Mistral-Nemo-Instruct-2407', 'Qwen/Qwen2.5-VL-32B-Instruct', 'meta-llama/Llama-3.3-70B-Instruct', 'mistralai/Devstral-Small-2505', 'mistralai/Magistral-Small-2506', 'mistralai/Mistral-Large-Instruct-2411', 'CohereForAI/aya-expanse-32b']
|
||||||
|
|
||||||
|
# Стандартный API ключ
|
||||||
|
DEFAULT_API_KEY=os.getenv('API_KEY')
|
||||||
|
# Задаем выходную директорию
|
||||||
|
OUTPUT_PATH='outputs'
|
||||||
|
|
||||||
|
# Стандартный системный промпт
|
||||||
|
OUTPUT_PATH='outputs'
|
||||||
|
|
||||||
|
DEFAULT_SYSTEM_PROMPT='''You are a diligent university student who has recorded a lecture as an audio file and later transcribed it into raw text.
|
||||||
|
Your task is to rewrite this unstructured transcript into a clear, logically organized, and detailed lecture summary (lecture notes).
|
||||||
|
|
||||||
|
Guidelines:
|
||||||
|
1. Structure:
|
||||||
|
- Organize the text into a hierarchy of sections and subsections.
|
||||||
|
- Use headings, bullet points, or numbering where appropriate.
|
||||||
|
- Present the material in a logical flow (from introduction → main points → details → examples → conclusion).
|
||||||
|
|
||||||
|
2. Clarity & Cohesion:
|
||||||
|
- Remove filler words, repetitions, and irrelevant fragments.
|
||||||
|
- Rewrite incomplete sentences into full, grammatically correct sentences.
|
||||||
|
- Ensure smooth transitions between topics, making the summary feel continuous and well-connected.
|
||||||
|
|
||||||
|
3. Depth & Detail:
|
||||||
|
- Capture all important concepts, definitions, examples, and explanations from the lecture.
|
||||||
|
- Expand shorthand or fragmented thoughts into full, precise explanations.
|
||||||
|
- Where appropriate, rephrase or clarify confusing passages for better understanding.
|
||||||
|
|
||||||
|
4. Accuracy:
|
||||||
|
- Preserve the lecturer’s original meaning, intent, and terminology.
|
||||||
|
- Avoid adding personal opinions or new information that was not in the lecture.
|
||||||
|
|
||||||
|
5. Style:
|
||||||
|
- Write in a formal, academic tone suitable for study notes.
|
||||||
|
- Aim for readability: concise sentences, but thorough coverage of concepts.
|
||||||
|
- Use emphasis (e.g., bold or italic text) only when it improves comprehension.
|
||||||
|
|
||||||
|
Final Output: A cohesive, detailed, and well-structured lecture summary, suitable for later studying and revision.
|
||||||
|
Use only russian language!
|
||||||
|
'''
|
||||||
|
|
||||||
0
services/__init__.py
Normal file
0
services/__init__.py
Normal file
39
services/fasterWhisper.py
Normal file
39
services/fasterWhisper.py
Normal file
@@ -0,0 +1,39 @@
|
|||||||
|
from faster_whisper import WhisperModel
|
||||||
|
|
||||||
|
class FasterWhisper:
|
||||||
|
def __init__(self):
|
||||||
|
pass
|
||||||
|
|
||||||
|
def recognize(self, model, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText):
|
||||||
|
model = WhisperModel(model, device='cuda', compute_type='float16') # Задаем модель
|
||||||
|
|
||||||
|
segments, _ = model.transcribe( # Распознаем текст
|
||||||
|
audioFile,
|
||||||
|
beam_size=beamSize,
|
||||||
|
vad_filter=vadFilter,
|
||||||
|
vad_parameters={
|
||||||
|
"min_silence_duration_ms": minSilenceDurationMs,
|
||||||
|
"speech_pad_ms": speechPadMs
|
||||||
|
},
|
||||||
|
temperature= [temp0, temp1, temp2],
|
||||||
|
word_timestamps=wordTimestamps,
|
||||||
|
no_speech_threshold=noSpeechThreshold,
|
||||||
|
condition_on_previous_text=conditionOnPreviousText
|
||||||
|
)
|
||||||
|
|
||||||
|
text = ''
|
||||||
|
|
||||||
|
for seg in segments:
|
||||||
|
text += f"[{self.format_timestamp(seg.start)} -> {self.format_timestamp(seg.end)}] {seg.text}" + '\n'
|
||||||
|
|
||||||
|
return(text)
|
||||||
|
|
||||||
|
def format_timestamp(self, seconds: float) -> str:
|
||||||
|
millis = int(seconds * 1000)
|
||||||
|
hours = millis // (3600 * 1000)
|
||||||
|
minutes = (millis % (3600 * 1000)) // (60 * 1000)
|
||||||
|
seconds_int = (millis % (60 * 1000)) // 1000
|
||||||
|
millis = millis % 1000
|
||||||
|
return f"{hours:02d}:{minutes:02d}:{seconds_int:02d},{millis:03d}"
|
||||||
|
|
||||||
|
|
||||||
Reference in New Issue
Block a user