This commit is contained in:
swrneko
2025-09-07 22:52:33 +03:00
parent b34d2fc22e
commit f37a8689df
5 changed files with 212 additions and 0 deletions

1
.env Normal file
View File

@@ -0,0 +1 @@
API_KEY=io-v2-eyJhbGciOiJSUzI1NiIsInR5cCI6IkpXVCJ9.eyJvd25lciI6ImU4OGE2MDE0LTVhMWQtNGY2OS05M2EyLTRkOGQ5NTMxYTE5NSIsImV4cCI6NDkxMDcwMTgxNH0.E-Ywo2jzcG3c8I7eRuufI5FuHFSOJN7AK2I65xqnsQ8qH_UuS8L6Fst6LSui6AVaZbAs_gpbNFUhe4GRhmgDbQ

166
app.py Normal file
View File

@@ -0,0 +1,166 @@
import gradio as gr
from faster_whisper import WhisperModel
from pathlib import Path
from dotenv import load_dotenv
import os
from services.llm import Llm
load_dotenv()
# Стандартный API ключ
DEFAULT_API_KEY=os.getenv('API_KEY')
# Задаем выходную директорию
OUTPUT_PATH='outputs'
# Стандартный системный промпт
DEFAULT_SYSTEM_PROMPT='''You are a diligent university student who has recorded a lecture as an audio file and later transcribed it into raw text.
Your task is to rewrite this unstructured transcript into a clear, logically organized, and detailed lecture summary (lecture notes).
Guidelines:
1. Structure:
- Organize the text into a hierarchy of sections and subsections.
- Use headings, bullet points, or numbering where appropriate.
- Present the material in a logical flow (from introduction → main points → details → examples → conclusion).
2. Clarity & Cohesion:
- Remove filler words, repetitions, and irrelevant fragments.
- Rewrite incomplete sentences into full, grammatically correct sentences.
- Ensure smooth transitions between topics, making the summary feel continuous and well-connected.
3. Depth & Detail:
- Capture all important concepts, definitions, examples, and explanations from the lecture.
- Expand shorthand or fragmented thoughts into full, precise explanations.
- Where appropriate, rephrase or clarify confusing passages for better understanding.
4. Accuracy:
- Preserve the lecturer’s original meaning, intent, and terminology.
- Avoid adding personal opinions or new information that was not in the lecture.
5. Style:
- Write in a formal, academic tone suitable for study notes.
- Aim for readability: concise sentences, but thorough coverage of concepts.
- Use emphasis (e.g., bold or italic text) only when it improves comprehension.
Final Output: A cohesive, detailed, and well-structured lecture summary, suitable for later studying and revision.
Use only russian language!
'''
faster_whisper_models_list = ['tiny', 'base', 'small', 'medium', 'large-v1', 'large-v2', 'large-v3', 'large', 'distil-large-v2', 'distil-large-v3', 'distil-large-v3.5', 'large-v3-turbo', 'turbo']
llm_models_list = ['openai/gpt-oss-120b', 'Qwen/Qwen3-235B-A22B-Thinking-2507', 'deepseek-ai/DeepSeek-R1-0528', 'meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8', 'openai/gpt-oss-20b', 'Intel/Qwen3-Coder-480B-A35B-Instruct-int4-mixed-ar', 'meta-llama/Llama-3.2-90B-Vision-Instruct', 'mistralai/Mistral-Nemo-Instruct-2407', 'Qwen/Qwen2.5-VL-32B-Instruct', 'meta-llama/Llama-3.3-70B-Instruct', 'mistralai/Devstral-Small-2505', 'mistralai/Magistral-Small-2506', 'mistralai/Mistral-Large-Instruct-2411', 'CohereForAI/aya-expanse-32b']
# Функция транскрибации
def recognize(model, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText):
model = WhisperModel(model, device='cuda', compute_type='float16') # Задаем модель
segments, _ = model.transcribe( # Распознаем текст
audioFile,
beam_size=beamSize,
vad_filter=vadFilter,
vad_parameters={
"min_silence_duration_ms": minSilenceDurationMs,
"speech_pad_ms": speechPadMs
},
temperature= [temp0, temp1, temp2],
word_timestamps=wordTimestamps,
no_speech_threshold=noSpeechThreshold,
condition_on_previous_text=conditionOnPreviousText
)
text = ''
for seg in segments:
text += f"[{format_timestamp(seg.start)} -> {format_timestamp(seg.end)}] {seg.text}" + '\n'
return(text)
def format_timestamp(seconds: float) -> str:
millis = int(seconds * 1000)
hours = millis // (3600 * 1000)
minutes = (millis % (3600 * 1000)) // (60 * 1000)
seconds_int = (millis % (60 * 1000)) // 1000
millis = millis % 1000
return f"{hours:02d}:{minutes:02d}:{seconds_int:02d},{millis:03d}"
# Функция сохранения файла
def saveFile(filename, text):
directory = Path(OUTPUT_PATH)
filePath = directory / filename
filePath.parent.mkdir(parents=True, exist_ok=True)
filePath.write_text(text, encoding='utf-8')
with gr.Blocks() as demo:
gr.HTML('''
<div align=center>
<h1>
Faster Whisper WebUI
</h1>
</div>
''')
with gr.Row():
with gr.Column():
# Колонка в левой части экрана с основным взаимодействием
with gr.Accordion(label='Recognization'):
with gr.Column():
audioFile = gr.Audio(label='Load audio for transcribe', type="filepath")
recognizeBtn = gr.Button('recognize', variant='primary')
with gr.Accordion(label='Recognized text'):
recognizedText = gr.TextArea(label='')
# Колонка в правой части экрана с настройками
with gr.Column():
with gr.Accordion(label='Settings'):
# Первое поле на всю ширину в акордионе настроек
filename = gr.Textbox(label='Output filename', value='output.txt')
# Акордион настроек faster whisper
with gr.Accordion(label='Faster whisper settings'):
with gr.Row():
# Левая колонка в акордионе
with gr.Column():
fastWhisperModel = gr.Dropdown(label='Model', choices=faster_whisper_models_list, interactive=True)
beamSize = gr.Number(label='beam_size', value=8, interactive=True)
noSpeechThreshold = gr.Number(label='no_speech_threshold', value=0.5, interactive=True)
vadFilter = gr.Checkbox(label='vad_filter', value=True, interactive=True)
wordTimestamps = gr.Checkbox(label='word_timestamps', value=True, interactive=True)
conditionOnPreviousText = gr.Checkbox(label='condition_on_previous_text', value=False, interactive=True)
# Правая колонка в акордионе
with gr.Column():
with gr.Accordion(label='Vad parameters'):
minSilenceDurationMs = gr.Number(label='min_silence_duration_ms', value=300, interactive=True)
speechPadMs = gr.Number(label='speech_pad_ms', value=200, interactive=True)
with gr.Accordion(label='Temperature'):
temp0 = gr.Number(label='temp_0', value=0.0, interactive=True)
temp1 = gr.Number(label='temp_1', value=0.2, interactive=True)
temp2 = gr.Number(label='temp_2', value=0.4, interactive=True)
# Нижний акордион настроек для api ключа llm
with gr.Accordion(label='ai.io.net api settings'):
apiKey = gr.Textbox(label='API key', value=DEFAULT_API_KEY, interactive=True)
with gr.Accordion(label='System prompt'):
systemPrompt = gr.Textbox(label='', value=DEFAULT_SYSTEM_PROMPT, interactive=True)
with gr.Row():
llmModel = gr.Dropdown(label='models', choices=llm_models_list, value=llm_models_list[0], interactive=True)
llmTemperature = gr.Number(label='Temperature', value=0.8, interactive=True )
with gr.Column():
with gr.Accordion(label='LLM'):
with gr.Accordion(label='Refined text raw'):
refinedText = gr.Textbox(label='')
with gr.Accordion(label='Refined text md formated'):
refinedTextMD = gr.Markdown(label='')
recognizeBtn.click(recognize, outputs=[recognizedText], inputs=[fastWhisperModel, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText])
recognizedText.change(Llm(apiKey.value).generate, inputs=[llmModel, systemPrompt, recognizedText, llmTemperature], outputs=[refinedText, refinedTextMD])
demo.launch()

14
outputs/output.txt Normal file
View File

@@ -0,0 +1,14 @@
[00:00:00,000 -> 00:00:03,680] Современные технологии стремительно принякают во всех свера нашей жизни.
[00:00:04,240 -> 00:00:08,340] Мы пользуемся смартфонами, интернетом, искусственным интеллектом и умными помощниками.
[00:00:09,000 -> 00:00:10,320] Мир меняется буквально на глазах.
[00:00:10,740 -> 00:00:13,460] Работа собирает автомобиля алгоритм и пишут статьи,
[00:00:13,640 -> 00:00:17,180] а голосую систему могут озвучивать текст с почти человеческой интернации.
[00:00:17,600 -> 00:00:21,860] Однако, несмотря на все достижение, важно помнить, технологии – это все в большин инструмент.
[00:00:22,500 -> 00:00:24,660] То, как мы их применяем зависит от нас.
[00:00:25,120 -> 00:00:29,080] Мы должны использовать прогресс-разумно, сохраняя человеческие ценности,
[00:00:29,080 -> 00:00:31,880] забудьтеся будущим и друг от друга.
[00:00:32,400 -> 00:00:34,660] Развитите технологий нет меняет важность доброты,
[00:00:34,940 -> 00:00:35,980] внимание и этики.
[00:00:36,220 -> 00:00:38,340] Пусть технологический прогресс делая жизнью
[00:00:38,340 -> 00:00:40,960] допни, но настоящую ценность предоёт и человек.
[00:00:41,200 -> 00:00:43,660] Все мыслими, чувствами и стремлянем к лучшему.

Binary file not shown.

31
services/llm.py Normal file
View File

@@ -0,0 +1,31 @@
import openai
class Llm:
def __init__(self, apiKey:str):
self.client = openai.OpenAI(
api_key=apiKey,
base_url='https://api.intelligence.io.solutions/api/v1/'
)
def generate(self, model:str, systemPrompt:str, userPrompt:str, temp:float):
'''
prompt[user_promtp, system_ptompt]
Function generate text by prompt
'''
# Получаем ответ от нейросети
response = self.client.chat.completions.create(
model=model,
messages=[
{'role': 'system', 'content': systemPrompt},
{'role': 'user', 'content': userPrompt},
],
temperature=temp,
stream=False
)
# Достаем текст
text = str(response.choices[0].message.content)
return text, text