change structure of the project

Changes:
   Move some modules for it's own file and move config to the `config.py`
This commit is contained in:
swrneko
2025-09-08 21:14:38 +03:00
parent 528704ab78
commit 00943f6525
4 changed files with 94 additions and 81 deletions

88
app.py
View File

@@ -1,79 +1,14 @@
import gradio as gr
from faster_whisper import WhisperModel
from pathlib import Path
from dotenv import load_dotenv
from typing import Optional
import os
# Подгрузка сервисов
from services.llm import Llm
from services.fasterWhisper import FasterWhisper
load_dotenv()
# Стандартный API ключ
DEFAULT_API_KEY=os.getenv('API_KEY')
# Задаем выходную директорию
OUTPUT_PATH='outputs'
# Стандартный системный промпт
DEFAULT_SYSTEM_PROMPT='''You are a diligent university student who has recorded a lecture as an audio file and later transcribed it into raw text.
Your task is to rewrite this unstructured transcript into a clear, logically organized, and detailed lecture summary (lecture notes).
Guidelines:
1. Structure:
- Organize the text into a hierarchy of sections and subsections.
- Use headings, bullet points, or numbering where appropriate.
- Present the material in a logical flow (from introduction → main points → details → examples → conclusion).
2. Clarity & Cohesion:
- Remove filler words, repetitions, and irrelevant fragments.
- Rewrite incomplete sentences into full, grammatically correct sentences.
- Ensure smooth transitions between topics, making the summary feel continuous and well-connected.
3. Depth & Detail:
- Capture all important concepts, definitions, examples, and explanations from the lecture.
- Expand shorthand or fragmented thoughts into full, precise explanations.
- Where appropriate, rephrase or clarify confusing passages for better understanding.
4. Accuracy:
- Preserve the lecturer’s original meaning, intent, and terminology.
- Avoid adding personal opinions or new information that was not in the lecture.
5. Style:
- Write in a formal, academic tone suitable for study notes.
- Aim for readability: concise sentences, but thorough coverage of concepts.
- Use emphasis (e.g., bold or italic text) only when it improves comprehension.
Final Output: A cohesive, detailed, and well-structured lecture summary, suitable for later studying and revision.
Use only russian language!
'''
faster_whisper_models_list = ['tiny', 'base', 'small', 'medium', 'large-v1', 'large-v2', 'large-v3', 'large', 'distil-large-v2', 'distil-large-v3', 'distil-large-v3.5', 'large-v3-turbo', 'turbo']
llm_models_list = ['openai/gpt-oss-120b', 'Qwen/Qwen3-235B-A22B-Thinking-2507', 'deepseek-ai/DeepSeek-R1-0528', 'meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8', 'openai/gpt-oss-20b', 'Intel/Qwen3-Coder-480B-A35B-Instruct-int4-mixed-ar', 'meta-llama/Llama-3.2-90B-Vision-Instruct', 'mistralai/Mistral-Nemo-Instruct-2407', 'Qwen/Qwen2.5-VL-32B-Instruct', 'meta-llama/Llama-3.3-70B-Instruct', 'mistralai/Devstral-Small-2505', 'mistralai/Magistral-Small-2506', 'mistralai/Mistral-Large-Instruct-2411', 'CohereForAI/aya-expanse-32b']
# Загрузка параметров конфигурации
from config import *
# Функция транскрибации
def recognize(model, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText):
model = WhisperModel(model, device='cuda', compute_type='float16') # Задаем модель
segments, _ = model.transcribe( # Распознаем текст
audioFile,
beam_size=beamSize,
vad_filter=vadFilter,
vad_parameters={
"min_silence_duration_ms": minSilenceDurationMs,
"speech_pad_ms": speechPadMs
},
temperature= [temp0, temp1, temp2],
word_timestamps=wordTimestamps,
no_speech_threshold=noSpeechThreshold,
condition_on_previous_text=conditionOnPreviousText
)
text = ''
for seg in segments:
text += f"[{format_timestamp(seg.start)} -> {format_timestamp(seg.end)}] {seg.text}" + '\n'
return(text)
def generateByCondition(api_key, llm_model, system_prompt, recognized_text, llm_temperature, is_pipeline_enabled, trigger, isSaveFile):
llm = Llm(api_key)
@@ -98,13 +33,6 @@ def generateByCondition(api_key, llm_model, system_prompt, recognized_text, llm_
# если нет чекбокса и было событие change
return gr.skip(), gr.skip()
def format_timestamp(seconds: float) -> str:
millis = int(seconds * 1000)
hours = millis // (3600 * 1000)
minutes = (millis % (3600 * 1000)) // (60 * 1000)
seconds_int = (millis % (60 * 1000)) // 1000
millis = millis % 1000
return f"{hours:02d}:{minutes:02d}:{seconds_int:02d},{millis:03d}"
# Функция сохранения файла
def saveFile(filename, text):
@@ -123,7 +51,6 @@ def updateButton(isChecked):
return gr.update(interactive=not isChecked, variant=variant)
def updateTextbox(isChecked):
return gr.update(visible=isChecked)
@@ -161,7 +88,7 @@ with gr.Blocks() as demo:
with gr.Row():
# Левая колонка в акордионе
with gr.Column():
fastWhisperModel = gr.Dropdown(label='Model', choices=faster_whisper_models_list, interactive=True)
fastWhisperModel = gr.Dropdown(label='Model', choices=FAST_WHISPER_MODELS, value=FAST_WHISPER_MODELS[11], interactive=True)
beamSize = gr.Number(label='beam_size', value=8, interactive=True)
noSpeechThreshold = gr.Number(label='no_speech_threshold', value=0.5, interactive=True)
@@ -188,7 +115,7 @@ with gr.Blocks() as demo:
systemPrompt = gr.Textbox(label='', value=DEFAULT_SYSTEM_PROMPT, interactive=True)
with gr.Row():
llmModel = gr.Dropdown(label='models', choices=llm_models_list, value=llm_models_list[0], interactive=True)
llmModel = gr.Dropdown(label='models', choices=LLM_MODELS, value=LLM_MODELS[0], interactive=True)
llmTemperature = gr.Number(label='Temperature', value=0.8, interactive=True )
with gr.Column():
@@ -206,7 +133,7 @@ with gr.Blocks() as demo:
isPipelineEnabledCheckbox.change(updateButton, inputs=[isPipelineEnabledCheckbox], outputs=refineTextBtn)
saveFileCheckbox.change(updateTextbox, inputs=saveFileCheckbox, outputs=filename)
recognizeBtn.click(recognize, outputs=[recognizedText], inputs=[fastWhisperModel, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText])
recognizeBtn.click(FasterWhisper.recognize, outputs=[recognizedText], inputs=[fastWhisperModel, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText])
# Если пайплайн включен то тогда делаем автоматически
# автоматический пайплайн
@@ -223,5 +150,4 @@ with gr.Blocks() as demo:
outputs=[refinedText, refinedTextMD]
)
demo.launch()