diff --git a/.env b/.env
new file mode 100644
index 0000000..7d75516
--- /dev/null
+++ b/.env
@@ -0,0 +1 @@
+API_KEY=io-v2-eyJhbGciOiJSUzI1NiIsInR5cCI6IkpXVCJ9.eyJvd25lciI6ImU4OGE2MDE0LTVhMWQtNGY2OS05M2EyLTRkOGQ5NTMxYTE5NSIsImV4cCI6NDkxMDcwMTgxNH0.E-Ywo2jzcG3c8I7eRuufI5FuHFSOJN7AK2I65xqnsQ8qH_UuS8L6Fst6LSui6AVaZbAs_gpbNFUhe4GRhmgDbQ
diff --git a/app.py b/app.py
new file mode 100644
index 0000000..4b15ab8
--- /dev/null
+++ b/app.py
@@ -0,0 +1,166 @@
+import gradio as gr
+from faster_whisper import WhisperModel
+from pathlib import Path
+from dotenv import load_dotenv
+import os
+
+from services.llm import Llm
+
+load_dotenv()
+
+# Стандартный API ключ
+DEFAULT_API_KEY=os.getenv('API_KEY')
+# Задаем выходную директорию
+OUTPUT_PATH='outputs'
+# Стандартный системный промпт
+DEFAULT_SYSTEM_PROMPT='''You are a diligent university student who has recorded a lecture as an audio file and later transcribed it into raw text.
+Your task is to rewrite this unstructured transcript into a clear, logically organized, and detailed lecture summary (lecture notes).
+
+Guidelines:
+1. Structure:
+ - Organize the text into a hierarchy of sections and subsections.
+ - Use headings, bullet points, or numbering where appropriate.
+ - Present the material in a logical flow (from introduction → main points → details → examples → conclusion).
+
+2. Clarity & Cohesion:
+ - Remove filler words, repetitions, and irrelevant fragments.
+ - Rewrite incomplete sentences into full, grammatically correct sentences.
+ - Ensure smooth transitions between topics, making the summary feel continuous and well-connected.
+
+3. Depth & Detail:
+ - Capture all important concepts, definitions, examples, and explanations from the lecture.
+ - Expand shorthand or fragmented thoughts into full, precise explanations.
+ - Where appropriate, rephrase or clarify confusing passages for better understanding.
+
+4. Accuracy:
+ - Preserve the lecturer’s original meaning, intent, and terminology.
+ - Avoid adding personal opinions or new information that was not in the lecture.
+
+5. Style:
+ - Write in a formal, academic tone suitable for study notes.
+ - Aim for readability: concise sentences, but thorough coverage of concepts.
+ - Use emphasis (e.g., bold or italic text) only when it improves comprehension.
+
+Final Output: A cohesive, detailed, and well-structured lecture summary, suitable for later studying and revision.
+Use only russian language!
+'''
+
+faster_whisper_models_list = ['tiny', 'base', 'small', 'medium', 'large-v1', 'large-v2', 'large-v3', 'large', 'distil-large-v2', 'distil-large-v3', 'distil-large-v3.5', 'large-v3-turbo', 'turbo']
+llm_models_list = ['openai/gpt-oss-120b', 'Qwen/Qwen3-235B-A22B-Thinking-2507', 'deepseek-ai/DeepSeek-R1-0528', 'meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8', 'openai/gpt-oss-20b', 'Intel/Qwen3-Coder-480B-A35B-Instruct-int4-mixed-ar', 'meta-llama/Llama-3.2-90B-Vision-Instruct', 'mistralai/Mistral-Nemo-Instruct-2407', 'Qwen/Qwen2.5-VL-32B-Instruct', 'meta-llama/Llama-3.3-70B-Instruct', 'mistralai/Devstral-Small-2505', 'mistralai/Magistral-Small-2506', 'mistralai/Mistral-Large-Instruct-2411', 'CohereForAI/aya-expanse-32b']
+
+
+# Функция транскрибации
+def recognize(model, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText):
+ model = WhisperModel(model, device='cuda', compute_type='float16') # Задаем модель
+
+ segments, _ = model.transcribe( # Распознаем текст
+ audioFile,
+ beam_size=beamSize,
+ vad_filter=vadFilter,
+ vad_parameters={
+ "min_silence_duration_ms": minSilenceDurationMs,
+ "speech_pad_ms": speechPadMs
+ },
+ temperature= [temp0, temp1, temp2],
+ word_timestamps=wordTimestamps,
+ no_speech_threshold=noSpeechThreshold,
+ condition_on_previous_text=conditionOnPreviousText
+ )
+
+ text = ''
+
+ for seg in segments:
+ text += f"[{format_timestamp(seg.start)} -> {format_timestamp(seg.end)}] {seg.text}" + '\n'
+
+ return(text)
+
+
+def format_timestamp(seconds: float) -> str:
+ millis = int(seconds * 1000)
+ hours = millis // (3600 * 1000)
+ minutes = (millis % (3600 * 1000)) // (60 * 1000)
+ seconds_int = (millis % (60 * 1000)) // 1000
+ millis = millis % 1000
+ return f"{hours:02d}:{minutes:02d}:{seconds_int:02d},{millis:03d}"
+
+# Функция сохранения файла
+def saveFile(filename, text):
+ directory = Path(OUTPUT_PATH)
+ filePath = directory / filename
+ filePath.parent.mkdir(parents=True, exist_ok=True)
+ filePath.write_text(text, encoding='utf-8')
+
+
+with gr.Blocks() as demo:
+ gr.HTML('''
+
+
+ Faster Whisper WebUI
+
+
+ ''')
+
+ with gr.Row():
+ with gr.Column():
+ # Колонка в левой части экрана с основным взаимодействием
+ with gr.Accordion(label='Recognization'):
+ with gr.Column():
+ audioFile = gr.Audio(label='Load audio for transcribe', type="filepath")
+ recognizeBtn = gr.Button('recognize', variant='primary')
+
+ with gr.Accordion(label='Recognized text'):
+ recognizedText = gr.TextArea(label='')
+
+ # Колонка в правой части экрана с настройками
+ with gr.Column():
+ with gr.Accordion(label='Settings'):
+ # Первое поле на всю ширину в акордионе настроек
+ filename = gr.Textbox(label='Output filename', value='output.txt')
+
+ # Акордион настроек faster whisper
+ with gr.Accordion(label='Faster whisper settings'):
+ with gr.Row():
+ # Левая колонка в акордионе
+ with gr.Column():
+ fastWhisperModel = gr.Dropdown(label='Model', choices=faster_whisper_models_list, interactive=True)
+
+ beamSize = gr.Number(label='beam_size', value=8, interactive=True)
+ noSpeechThreshold = gr.Number(label='no_speech_threshold', value=0.5, interactive=True)
+ vadFilter = gr.Checkbox(label='vad_filter', value=True, interactive=True)
+ wordTimestamps = gr.Checkbox(label='word_timestamps', value=True, interactive=True)
+ conditionOnPreviousText = gr.Checkbox(label='condition_on_previous_text', value=False, interactive=True)
+
+ # Правая колонка в акордионе
+ with gr.Column():
+ with gr.Accordion(label='Vad parameters'):
+ minSilenceDurationMs = gr.Number(label='min_silence_duration_ms', value=300, interactive=True)
+ speechPadMs = gr.Number(label='speech_pad_ms', value=200, interactive=True)
+
+ with gr.Accordion(label='Temperature'):
+ temp0 = gr.Number(label='temp_0', value=0.0, interactive=True)
+ temp1 = gr.Number(label='temp_1', value=0.2, interactive=True)
+ temp2 = gr.Number(label='temp_2', value=0.4, interactive=True)
+
+ # Нижний акордион настроек для api ключа llm
+ with gr.Accordion(label='ai.io.net api settings'):
+ apiKey = gr.Textbox(label='API key', value=DEFAULT_API_KEY, interactive=True)
+
+ with gr.Accordion(label='System prompt'):
+ systemPrompt = gr.Textbox(label='', value=DEFAULT_SYSTEM_PROMPT, interactive=True)
+
+ with gr.Row():
+ llmModel = gr.Dropdown(label='models', choices=llm_models_list, value=llm_models_list[0], interactive=True)
+ llmTemperature = gr.Number(label='Temperature', value=0.8, interactive=True )
+ with gr.Column():
+ with gr.Accordion(label='LLM'):
+ with gr.Accordion(label='Refined text raw'):
+ refinedText = gr.Textbox(label='')
+
+ with gr.Accordion(label='Refined text md formated'):
+ refinedTextMD = gr.Markdown(label='')
+
+
+ recognizeBtn.click(recognize, outputs=[recognizedText], inputs=[fastWhisperModel, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText])
+ recognizedText.change(Llm(apiKey.value).generate, inputs=[llmModel, systemPrompt, recognizedText, llmTemperature], outputs=[refinedText, refinedTextMD])
+
+demo.launch()
diff --git a/outputs/output.txt b/outputs/output.txt
new file mode 100644
index 0000000..be45db8
--- /dev/null
+++ b/outputs/output.txt
@@ -0,0 +1,14 @@
+[00:00:00,000 -> 00:00:03,680] Современные технологии стремительно принякают во всех свера нашей жизни.
+[00:00:04,240 -> 00:00:08,340] Мы пользуемся смартфонами, интернетом, искусственным интеллектом и умными помощниками.
+[00:00:09,000 -> 00:00:10,320] Мир меняется буквально на глазах.
+[00:00:10,740 -> 00:00:13,460] Работа собирает автомобиля алгоритм и пишут статьи,
+[00:00:13,640 -> 00:00:17,180] а голосую систему могут озвучивать текст с почти человеческой интернации.
+[00:00:17,600 -> 00:00:21,860] Однако, несмотря на все достижение, важно помнить, технологии – это все в большин инструмент.
+[00:00:22,500 -> 00:00:24,660] То, как мы их применяем зависит от нас.
+[00:00:25,120 -> 00:00:29,080] Мы должны использовать прогресс-разумно, сохраняя человеческие ценности,
+[00:00:29,080 -> 00:00:31,880] забудьтеся будущим и друг от друга.
+[00:00:32,400 -> 00:00:34,660] Развитите технологий нет меняет важность доброты,
+[00:00:34,940 -> 00:00:35,980] внимание и этики.
+[00:00:36,220 -> 00:00:38,340] Пусть технологический прогресс делая жизнью
+[00:00:38,340 -> 00:00:40,960] допни, но настоящую ценность предоёт и человек.
+[00:00:41,200 -> 00:00:43,660] Все мыслими, чувствами и стремлянем к лучшему.
diff --git a/services/__pycache__/llm.cpython-310.pyc b/services/__pycache__/llm.cpython-310.pyc
new file mode 100644
index 0000000..1651e1c
Binary files /dev/null and b/services/__pycache__/llm.cpython-310.pyc differ
diff --git a/services/llm.py b/services/llm.py
new file mode 100644
index 0000000..1ebc57b
--- /dev/null
+++ b/services/llm.py
@@ -0,0 +1,31 @@
+import openai
+
+class Llm:
+ def __init__(self, apiKey:str):
+ self.client = openai.OpenAI(
+ api_key=apiKey,
+ base_url='https://api.intelligence.io.solutions/api/v1/'
+ )
+
+ def generate(self, model:str, systemPrompt:str, userPrompt:str, temp:float):
+ '''
+ prompt[user_promtp, system_ptompt]
+ Function generate text by prompt
+ '''
+
+ # Получаем ответ от нейросети
+ response = self.client.chat.completions.create(
+ model=model,
+ messages=[
+ {'role': 'system', 'content': systemPrompt},
+ {'role': 'user', 'content': userPrompt},
+ ],
+ temperature=temp,
+ stream=False
+ )
+
+ # Достаем текст
+ text = str(response.choices[0].message.content)
+
+ return text, text
+