1 Commits
dev ... swrneko

Author SHA1 Message Date
Voskik
bb9ac27d85 Update .gitignore 2025-09-13 01:50:17 +03:00
19 changed files with 1074 additions and 1526 deletions

9
.gitignore vendored
View File

@@ -1,5 +1,4 @@
.env
__pycache__/
outputs/*
venv/
.env
__pycache__/
outputs/*
venv/

1348
LICENSE

File diff suppressed because it is too large Load Diff

118
README.md
View File

@@ -1,59 +1,59 @@
# FWAL WebUI (Faster Whisper And LLM WebUI) by swrneko
<div align="center">
<img src="https://count.getloli.com/get/@swrneko-faster-whisper-llm?theme=rule34"/>
</div>
## Screenshots
<div align="center">
<div>
<img src="src/1.png" style="object-fit: cover;"/>
</div>
<div style="align-items: center;">
<img src="src/2.png" style="width: 49.7%; object-fit: cover;"/>
<img src="src/3.png" style="width: 49.7%; object-fit: cover;"/>
</div>
</div>
## Requirements
- python-conda or miniconda;
- python 3.10 or above;
- linux (windows not tested but probably working);
Python requirements are listed in `requirements.txt`.
## Installation
1. Clone repository:
```
git clone https://github.com/swrneko/faster-whisper-n-ionet-llm.git
cd faster-whisper-n-ionet-llm
```
also (if not insatlled)
- Insatll conda:
```
wget https://repo.anaconda.com/miniconda/Miniconda3-latest-Linux-x86_64.sh -O miniconda.sh && bash miniconda.sh
```
2. Create virtual env:
```
conda create -n faster-whisper-n-ionet-llm python=3.10
conda activate faster-whisper-n-ionet-llm
conda install nvidia::cudnn cuda-version=12
```
3. Install requirements:
```
pip install -r requirements.txt
```
4. Get api key from [io.net](https://ai.io.net/ai/api-keys) and insert into `.env` file (need to create it in root of repository directory).
It should looks like this:
```
API_KEY='your_api_key_without_qoutes'
```
5. Done! Now you can just run it like that:
```shell
python app.py
```
# FWAL WebUI (Faster Whisper And LLM WebUI) by swrneko
<div align="center">
<img src="https://count.getloli.com/get/@swrneko-faster-whisper-llm?theme=rule34"/>
</div>
## Screenshots
<div align="center">
<div>
<img src="src/1.png" style="object-fit: cover;"/>
</div>
<div style="align-items: center;">
<img src="src/2.png" style="width: 49.7%; object-fit: cover;"/>
<img src="src/3.png" style="width: 49.7%; object-fit: cover;"/>
</div>
</div>
## Requirements
- python-conda or miniconda;
- python 3.10 or above;
- linux (windows not tested but probably working);
Python requirements are listed in `requirements.txt`.
## Installation
1. Clone repository:
```
git clone https://github.com/swrneko/faster-whisper-n-ionet-llm.git
cd faster-whisper-n-ionet-llm
```
also (if not insatlled)
- Insatll conda:
```
wget https://repo.anaconda.com/miniconda/Miniconda3-latest-Linux-x86_64.sh -O miniconda.sh && bash miniconda.sh
```
2. Create virtual env:
```
conda create -n faster-whisper-n-ionet-llm python=3.10
conda activate faster-whisper-n-ionet-llm
conda install nvidia::cudnn cuda-version=12
```
3. Install requirements:
```
pip install -r requirements.txt
```
4. Get api key from [io.net](https://ai.io.net/ai/api-keys) and insert into `.env` file (need to create it in root of repository directory).
It should looks like this:
```
API_KEY='your_api_key_without_qoutes'
```
5. Done! Now you can just run it like that:
```shell
python app.py
```

346
app.py
View File

@@ -1,151 +1,195 @@
import gradio as gr
# Загрузка параметров конфигурации
from config import *
# Подгрузка сервисов
from services.llm_factory import get_llm_provider
from services.fasterWhisper import FasterWhisper
# Загрузка доп. модулей
from handlers.gradioHandler import GradioHandlers
from handlers.fileHandlers import FileHandlers
from handlers.convertMdToPdf import ConvertMdToPdf
from handlers.glueAudio import GlueAudio
gh = GradioHandlers(get_llm_provider, ConvertMdToPdf, FileHandlers, FasterWhisper, GlueAudio)
def main():
with gr.Blocks() as demo:
gr.HTML('''
<div align=center>
<h1>
Faster Whisper WebUI
</h1>
</div>
''')
with gr.Row():
# Вкладка с основным взаимодействием
with gr.Tab('Actions'):
isPipelineEnabledCheckbox = gr.Checkbox(label='is pipeline enabled', value=True, interactive=True)
with gr.Row():
with gr.Accordion(label='Recognization and integration'):
with gr.Column():
audioFiles = gr.Files(label='Load audio for transcribe', type="filepath")
images = gr.Files(label='Upload images', file_types=['image'])
recognizeBtn = gr.Button('recognize and integrate', variant='primary')
with gr.Accordion(label='Recognized text'):
recognizedText = gr.TextArea(label='')
with gr.Accordion(label='LLM'):
with gr.Column():
refineTextBtn = gr.Button('refine text', variant='secondary', interactive=False)
with gr.Accordion(label='Refined text raw'):
refinedText = gr.Textbox(label='', show_copy_button=True)
with gr.Accordion(label='Refined text md formated'):
refinedTextMD = gr.Markdown(label='')
# Вкладка с настройками
with gr.Tab('Settings'):
with gr.Column():
# Первое поле на всю ширину в акордионе настроек
with gr.Accordion('File settings'):
saveFileCheckbox = gr.Checkbox(label='save file', value=True, interactive=True)
filename = gr.Textbox(label='Output filename', value='output.txt', interactive=True)
filenamePdf = gr.Textbox(label='Output filename for pdf', value='output.pdf', interactive=True)
# Акордион настроек faster whisper
with gr.Accordion(label='Faster whisper settings'):
with gr.Row():
# Левая колонка в акордионе
with gr.Column():
device = gr.Dropdown(label='Device', choices=DEVICES, value=DEVICES[1], interactive=True)
compute_type = gr.Dropdown(label='compute_type', choices=COMPUTE_TYPE, value=COMPUTE_TYPE[0], interactive=True)
fastWhisperModel = gr.Dropdown(label='Model', choices=FAST_WHISPER_MODELS, value=FAST_WHISPER_MODELS[11], interactive=True)
beamSize = gr.Number(label='beam_size', value=8, interactive=True)
noSpeechThreshold = gr.Number(label='no_speech_threshold', value=0.5, interactive=True)
vadFilter = gr.Checkbox(label='vad_filter', value=True, interactive=True)
wordTimestamps = gr.Checkbox(label='word_timestamps', value=True, interactive=True)
conditionOnPreviousText = gr.Checkbox(label='condition_on_previous_text', value=False, interactive=True)
# Правая колонка в акордионе
with gr.Column():
with gr.Accordion(label='Vad parameters'):
minSilenceDurationMs = gr.Number(label='min_silence_duration_ms', value=300, interactive=True)
speechPadMs = gr.Number(label='speech_pad_ms', value=200, interactive=True)
with gr.Accordion(label='Temperature'):
temp0 = gr.Number(label='temp_0', value=0.0, interactive=True)
temp1 = gr.Number(label='temp_1', value=0.2, interactive=True)
temp2 = gr.Number(label='temp_2', value=0.4, interactive=True)
# Нижний акордион настроек для api ключа llm
with gr.Accordion(label='LLM settings'):
apiKey = gr.Textbox(label='API key (required for io.net, Gemini)', value=DEFAULT_API_KEY, interactive=True)
with gr.Accordion(label='System prompt'):
systemPrompt = gr.Textbox(label='', value=DEFAULT_SYSTEM_PROMPT, interactive=True)
with gr.Row():
# ВЫБОР ПРОВАЙДЕРА
llmProvider = gr.Dropdown(
label='LLM Provider',
choices=LLM_PROVIDERS,
value=LLM_PROVIDERS[0],
interactive=True
)
# СПИСОК МОДЕЛЕЙ (теперь зависит от провайдера)
llmModel = gr.Dropdown(
label='Models',
choices=LLM_MODELS[LLM_PROVIDERS[0]], # Модели для провайдера по умолчанию
value=LLM_MODELS[LLM_PROVIDERS[0]][1],
interactive=True
)
llmTemperature = gr.Number(label='Temperature', value=0.8, interactive=True)
isPipelineEnabledCheckbox.change(gh.updateButton, inputs=[isPipelineEnabledCheckbox], outputs=refineTextBtn)
saveFileCheckbox.change(gh.updateTextbox, inputs=saveFileCheckbox, outputs=filename)
saveFileCheckbox.change(gh.updateTextbox, inputs=saveFileCheckbox, outputs=filenamePdf)
recognizeBtn.click(
gh.handleRecognizeBtn,
inputs=[audioFiles, fastWhisperModel, device, compute_type, beamSize,
vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2,
wordTimestamps, noSpeechThreshold, conditionOnPreviousText, gr.State(GLUED_AUDIO_FILENAME), gr.State(OUTPUT_PATH)],
outputs=[recognizedText],
)
# Если пайплайн включен то тогда делаем автоматически
# автоматический пайплайн
recognizedText.change(
gh.generateByCondition,
inputs=[apiKey, llmProvider, llmModel, systemPrompt, recognizedText, llmTemperature,
isPipelineEnabledCheckbox, gr.State("change"), saveFileCheckbox, filename, filenamePdf, gr.State(OUTPUT_PATH)],
outputs=[refinedText, refinedTextMD]
)
# ручной запуск по кнопке
llmProvider.change(
gh.update_model_dropdown,
inputs=llmProvider,
outputs=[llmModel, apiKey]
)
refineTextBtn.click(
gh.generateByCondition,
inputs=[apiKey, llmProvider, llmModel, systemPrompt, recognizedText, llmTemperature,
isPipelineEnabledCheckbox, gr.State("click"), saveFileCheckbox, filename, filenamePdf, gr.State(OUTPUT_PATH)],
outputs=[refinedText, refinedTextMD]
)
demo.launch()
if __name__ == '__main__':
main()
import gradio as gr
from pathlib import Path
# Подгрузка сервисов
from services.llm import Llm
from services.fasterWhisper import FasterWhisper
from services.convertMdToPdf import ConvertMdToPdf
# Загрузка параметров конфигурации
from config import *
# Функция транскрибации
def generateByCondition(api_key, llm_model, system_prompt, recognized_text, llm_temperature, is_pipeline_enabled, trigger, isSaveFile, filename, filenamePdf):
llm = Llm(api_key)
# если чекбокс включен и событие было change → обрабатываем
if is_pipeline_enabled and trigger == "change":
result, md = llm.generate(llm_model, system_prompt, recognized_text, llm_temperature)
# Конвертируем текст с латексом в юникод
pdf, unicodeText = ConvertMdToPdf().convertLatexToText(md)
if isSaveFile:
savePdf(filenamePdf, pdf)
saveFile(filename, result)
return result, unicodeText
# если чекбокс выключен и событие было click → обрабатываем
if not is_pipeline_enabled and trigger == "click":
result, md = llm.generate(llm_model, system_prompt, recognized_text, llm_temperature)
# Конвертируем текст с латексом в юникод
pdf, unicodeText = ConvertMdToPdf().convertLatexToText(md)
if isSaveFile:
savePdf(filenamePdf, pdf)
saveFile(filename, result)
return result, unicodeText
# если нет чекбокса и было событие change
return gr.skip(), gr.skip()
def savePdf(filename, pdf):
directory = Path(OUTPUT_PATH)
filePath = directory / filename
filePath.parent.mkdir(parents=True, exist_ok=True)
pdf.save(filePath)
# Функция сохранеhния файла
def saveFile(filename, text):
directory = Path(OUTPUT_PATH)
filePath = directory / filename
filePath.parent.mkdir(parents=True, exist_ok=True)
filePath.write_text(text, encoding='utf-8')
# ConvertMdToPdf().convert(text)
# Функция для динамического обновления кнопки в зависимости от состояния checkbox
def updateButton(isChecked):
if not isChecked:
variant = 'primary'
else:
variant = 'secondary'
return gr.update(interactive=not isChecked, variant=variant)
def updateTextbox(isChecked):
return gr.update(visible=isChecked)
###################################################
# ____ ___.___ ___. .__ #
#| | \ | \_ |__ ____ | | ______ _ __#
#| | / | | __ \_/ __ \| | / _ \ \/ \/ /#
#| | /| | | \_\ \ ___/| |_( <_> ) / #
#|______/ |___| |___ /\___ >____/\____/ \/\_/ #
# \/ \/ #
###################################################
with gr.Blocks() as demo:
gr.HTML('''
<div align=center>
<h1>
Faster Whisper WebUI
</h1>
</div>
''')
with gr.Row():
# Вкладка с основным взаимодействием
with gr.Tab('Actions'):
isPipelineEnabledCheckbox = gr.Checkbox(label='is pipeline enabled', value=True, interactive=True)
with gr.Row():
with gr.Accordion(label='Recognization and integration'):
with gr.Column():
audioFile = gr.Audio(label='Load audio for transcribe', type="filepath")
images = gr.Files(label='Upload images', file_types=['image'])
recognizeBtn = gr.Button('recognize and integrate', variant='primary')
with gr.Accordion(label='Recognized text'):
recognizedText = gr.TextArea(label='')
with gr.Accordion(label='LLM'):
with gr.Column():
refineTextBtn = gr.Button('refine text', variant='secondary', interactive=False)
with gr.Accordion(label='Refined text raw'):
refinedText = gr.Textbox(label='', show_copy_button=True)
with gr.Accordion(label='Refined text md formated'):
refinedTextMD = gr.Markdown(label='')
# Вкладка с настройками
with gr.Tab('Settings'):
with gr.Column():
# Первое поле на всю ширину в акордионе настроек
with gr.Accordion('File settings'):
saveFileCheckbox = gr.Checkbox(label='save file', value=True, interactive=True)
filename = gr.Textbox(label='Output filename', value='output.txt', interactive=True)
filenamePdf = gr.Textbox(label='Output filename for pdf', value='output.pdf', interactive=True)
# Акордион настроек faster whisper
with gr.Accordion(label='Faster whisper settings'):
with gr.Row():
# Левая колонка в акордионе
with gr.Column():
device = gr.Dropdown(label='Device', choices=["cpu", "cuda"], value="cuda", interactive=True)
compute_type = gr.Dropdown(label='compute_type', choices=["auto", "int8", "float16", "float32"], value="auto", interactive=True)
fastWhisperModel = gr.Dropdown(label='Model', choices=FAST_WHISPER_MODELS, value=FAST_WHISPER_MODELS[11], interactive=True)
beamSize = gr.Number(label='beam_size', value=8, interactive=True)
noSpeechThreshold = gr.Number(label='no_speech_threshold', value=0.5, interactive=True)
vadFilter = gr.Checkbox(label='vad_filter', value=True, interactive=True)
wordTimestamps = gr.Checkbox(label='word_timestamps', value=True, interactive=True)
conditionOnPreviousText = gr.Checkbox(label='condition_on_previous_text', value=False, interactive=True)
# Правая колонка в акордионе
with gr.Column():
with gr.Accordion(label='Vad parameters'):
minSilenceDurationMs = gr.Number(label='min_silence_duration_ms', value=300, interactive=True)
speechPadMs = gr.Number(label='speech_pad_ms', value=200, interactive=True)
with gr.Accordion(label='Temperature'):
temp0 = gr.Number(label='temp_0', value=0.0, interactive=True)
temp1 = gr.Number(label='temp_1', value=0.2, interactive=True)
temp2 = gr.Number(label='temp_2', value=0.4, interactive=True)
# Нижний акордион настроек для api ключа llm
with gr.Accordion(label='ai.io.net api settings'):
apiKey = gr.Textbox(label='API key', value=DEFAULT_API_KEY, interactive=True)
with gr.Accordion(label='System prompt'):
systemPrompt = gr.Textbox(label='', value=DEFAULT_SYSTEM_PROMPT, interactive=True)
with gr.Row():
llmModel = gr.Dropdown(label='models', choices=LLM_MODELS, value=LLM_MODELS[1], interactive=True)
llmTemperature = gr.Number(label='Temperature', value=0.8, interactive=True )
######################################################################
#.____ .__ ___. .__ #
#| | ____ ____ |__| ____ \_ |__ ____ | | ______ _ __#
#| | / _ \ / ___\| |/ ___\ | __ \_/ __ \| | / _ \ \/ \/ /#
#| |__( <_> ) /_/ > \ \___ | \_\ \ ___/| |_( <_> ) / #
#|_______ \____/\___ /|__|\___ > |___ /\___ >____/\____/ \/\_/ #
# \/ /_____/ \/ \/ \/ #
######################################################################
isPipelineEnabledCheckbox.change(updateButton, inputs=[isPipelineEnabledCheckbox], outputs=refineTextBtn)
saveFileCheckbox.change(updateTextbox, inputs=saveFileCheckbox, outputs=filename)
saveFileCheckbox.change(updateTextbox, inputs=saveFileCheckbox, outputs=filenamePdf)
recognizeBtn.click(FasterWhisper().recognize, outputs=[recognizedText], inputs=[fastWhisperModel, device, compute_type, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText])
# Если пайплайн включен то тогда делаем автоматически
# автоматический пайплайн
recognizedText.change(
generateByCondition,
inputs=[apiKey, llmModel, systemPrompt, recognizedText, llmTemperature, isPipelineEnabledCheckbox, gr.State("change"), saveFileCheckbox, filename, filenamePdf],
outputs=[refinedText, refinedTextMD]
)
# ручной запуск по кнопке
refineTextBtn.click(
generateByCondition,
inputs=[apiKey, llmModel, systemPrompt, recognizedText, llmTemperature, isPipelineEnabledCheckbox, gr.State("click"), saveFileCheckbox, filename, filenamePdf],
outputs=[refinedText, refinedTextMD]
)
demo.launch()

134
config.py
View File

@@ -1,86 +1,48 @@
import os
from dotenv import load_dotenv
load_dotenv()
FAST_WHISPER_MODELS = ['tiny', 'base', 'small', 'medium', 'large-v1', 'large-v2', 'large-v3', 'large', 'distil-large-v2', 'distil-large-v3', 'distil-large-v3.5', 'large-v3-turbo', 'turbo']
DEVICES = ['cpu', 'cuda']
COMPUTE_TYPE = ['auto', 'int8', 'float16', 'float32']
# Стандартный API ключ
IO_API_KEY=os.getenv('IO_API_KEY')
GEMINI_API_KEY=os.getenv('GEMINI_API_KEY')
DEFAULT_API_KEY=IO_API_KEY
# Словарь провайдеров и их моделей
LLM_PROVIDERS = ['io.net', 'Gemini', 'gpt4free']
LLM_MODELS = {
'io.net': [
'openai/gpt-oss-120b', 'Qwen/Qwen3-235B-A22B-Thinking-2507',
'deepseek-ai/DeepSeek-R1-0528', 'meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8',
'openai/gpt-oss-20b', 'Intel/Qwen3-Coder-480B-A35B-Instruct-int4-mixed-ar',
'meta-llama/Llama-3.2-90B-Vision-Instruct', 'mistralai/Mistral-Nemo-Instruct-2407',
'Qwen/Qwen2.5-VL-32B-Instruct', 'meta-llama/Llama-3.3-70B-Instruct',
'mistralai/Devstral-Small-2505', 'mistralai/Magistral-Small-2506',
'mistralai/Mistral-Large-Instruct-2411', 'CohereForAI/aya-expanse-32b'
],
'Gemini': [
'gemini-2.5-pro',
'gemini-2.5-flash',
'gemini-2.5-flash-lite'
],
'gpt4free': [ # Модели могут меняться, проверьте документацию g4f
'default',
'gpt-4',
'sonar-reasoning',
'command-r-plus',
'llama-3.3-70b',
'hermes-3-llama-3.1-405b'
'qwen-3-235b',
'gpt-4o-mini',
'deepseek-r1',
'PollinationsAI:gpt-5-nano'
]
}
# Задаем выходную директорию
OUTPUT_PATH='outputs'
GLUED_AUDIO_FILENAME='glued.mp3'
DEFAULT_SYSTEM_PROMPT='''You are a diligent university student who has recorded a lecture as an audio file and later transcribed it into raw text.
Your task is to rewrite this unstructured transcript into a clear, logically organized, and detailed lecture summary (lecture notes).
Guidelines:
1. Structure:
- Organize the text into a hierarchy of sections and subsections.
- Use headings, bullet points, or numbering where appropriate.
- Present the material in a logical flow (from introduction → main points → details → examples → conclusion).
2. Clarity & Cohesion:
- Remove filler words, repetitions, and irrelevant fragments.
- Rewrite incomplete sentences into full, grammatically correct sentences.
- Ensure smooth transitions between topics, making the summary feel continuous and well-connected.
3. Depth & Detail:
- Capture all important concepts, definitions, examples, and explanations from the lecture.
- Expand shorthand or fragmented thoughts into full, precise explanations.
- Where appropriate, rephrase or clarify confusing passages for better understanding.
4. Accuracy:
- Preserve the lecturer’s original meaning, intent, and terminology.
- Avoid adding personal opinions or new information that was not in the lecture.
5. Style:
- Write in a formal, academic tone suitable for study notes.
- Aim for readability: concise sentences, but thorough coverage of concepts.
- Use emphasis (e.g., bold or italic text) only when it improves comprehension.
Final Output: A cohesive, detailed, and well-structured lecture summary, suitable for later studying and revision.
Use only russian language!
USE LATEX IN DOLLAR SIGN ($)!
EXTRA BIG LENTH OF CONSPECT!
MAKE AS LONG AS POSIBLE AND AS BE GOOD!
'''
import os
from dotenv import load_dotenv
load_dotenv()
FAST_WHISPER_MODELS = ['tiny', 'base', 'small', 'medium', 'large-v1', 'large-v2', 'large-v3', 'large', 'distil-large-v2', 'distil-large-v3', 'distil-large-v3.5', 'large-v3-turbo', 'turbo']
LLM_MODELS = ['openai/gpt-oss-120b', 'Qwen/Qwen3-235B-A22B-Thinking-2507', 'deepseek-ai/DeepSeek-R1-0528', 'meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8', 'openai/gpt-oss-20b', 'Intel/Qwen3-Coder-480B-A35B-Instruct-int4-mixed-ar', 'meta-llama/Llama-3.2-90B-Vision-Instruct', 'mistralai/Mistral-Nemo-Instruct-2407', 'Qwen/Qwen2.5-VL-32B-Instruct', 'meta-llama/Llama-3.3-70B-Instruct', 'mistralai/Devstral-Small-2505', 'mistralai/Magistral-Small-2506', 'mistralai/Mistral-Large-Instruct-2411', 'CohereForAI/aya-expanse-32b']
# Стандартный API ключ
DEFAULT_API_KEY=os.getenv('API_KEY')
# Задаем выходную директорию
OUTPUT_PATH='outputs'
DEFAULT_SYSTEM_PROMPT='''You are a diligent university student who has recorded a lecture as an audio file and later transcribed it into raw text.
Your task is to rewrite this unstructured transcript into a clear, logically organized, and detailed lecture summary (lecture notes).
Guidelines:
1. Structure:
- Organize the text into a hierarchy of sections and subsections.
- Use headings, bullet points, or numbering where appropriate.
- Present the material in a logical flow (from introduction → main points → details → examples → conclusion).
2. Clarity & Cohesion:
- Remove filler words, repetitions, and irrelevant fragments.
- Rewrite incomplete sentences into full, grammatically correct sentences.
- Ensure smooth transitions between topics, making the summary feel continuous and well-connected.
3. Depth & Detail:
- Capture all important concepts, definitions, examples, and explanations from the lecture.
- Expand shorthand or fragmented thoughts into full, precise explanations.
- Where appropriate, rephrase or clarify confusing passages for better understanding.
4. Accuracy:
- Preserve the lecturer’s original meaning, intent, and terminology.
- Avoid adding personal opinions or new information that was not in the lecture.
5. Style:
- Write in a formal, academic tone suitable for study notes.
- Aim for readability: concise sentences, but thorough coverage of concepts.
- Use emphasis (e.g., bold or italic text) only when it improves comprehension.
Final Output: A cohesive, detailed, and well-structured lecture summary, suitable for later studying and revision.
Use only russian language!
USE LATEX IN DOLLAR SIGN ($)!
EXTRA BIG LENTH OF CONSPECT!
'''

View File

View File

@@ -1,31 +0,0 @@
from pathlib import Path
# Для аннотации типов
from markdown_pdf import MarkdownPdf
from pydub import AudioSegment
class FileHandlers:
# Функция сохранения файла
def saveFile(self, filename, content, output_path, format='mp3'):
'''
Сохраняет текст, pdf из markdown_pdf или склеенный аудиофайл в файл с указанным названием и директорией.
Args:
:param filename: название файла;
:param content: содержание файла;
:param output_path: выходная диретория файла.
'''
# Создание объекта директории
directory = Path(output_path)
filePath = directory / filename # Добавление пути директории
filePath.parent.mkdir(parents=True, exist_ok=True) # Создание директории если не существует
# Сохранение для разных типов
if type(content) == MarkdownPdf:
return content.save(filePath)
elif type(content) == str:
return filePath.write_text(content, encoding='utf-8')
elif type(content) == AudioSegment:
return content.export(filePath, format=format)

View File

@@ -1,61 +0,0 @@
import subprocess
from pathlib import Path
class GlueAudio():
def glue(self, audio_files: list, output_path: str, output_filename: str) -> Path:
"""
Склеивает аудиофайлы с помощью FFmpeg, используя промежуточный список файлов.
Этот метод чрезвычайно эффективен по памяти и скорости.
Args:
audio_files (list): Список путей к исходным аудиофайлам.
output_path (str): Директория для сохранения итогового файла.
output_filename (str): Имя итогового склеенного файла.
Returns:
Path: Путь к созданному склеенному файлу.
"""
output_dir = Path(output_path)
output_dir.mkdir(parents=True, exist_ok=True)
final_audio_path = output_dir / output_filename
if not audio_files:
raise ValueError("Список аудиофайлов для склейки пуст.")
# 1. Формируем часть команды с входными файлами (-i file1 -i file2 ...)
input_args = []
for file_path in audio_files:
input_args.extend(['-i', str(Path(file_path).resolve())])
# 2. Формируем строку для filter_complex
num_files = len(audio_files)
stream_specifiers = "".join([f"[{i}:a]" for i in range(num_files)])
filter_complex_str = f"{stream_specifiers}concat=n={num_files}:v=0:a=1[outa]"
# 3. Собираем полную команду
command = [
'ffmpeg',
*input_args, # Распаковываем список входных файлов
'-filter_complex', filter_complex_str,
'-map', '[outa]',
'-c:a', 'libmp3lame',
'-q:a', '2',
str(final_audio_path),
'-y'
]
try:
# 4. Выполняем команду
print(f"Выполнение команды FFmpeg: {' '.join(command)}")
subprocess.run(command, check=True, capture_output=True, text=True)
print("FFmpeg успешно завершил склейку.")
except FileNotFoundError:
raise FileNotFoundError("FFmpeg не найден. Убедитесь, что он установлен и доступен в системной переменной PATH.")
except subprocess.CalledProcessError as e:
print("Ошибка при выполнении FFmpeg!")
print("Stderr:", e.stderr)
raise RuntimeError(f"Ошибка FFmpeg при склейке файлов: {e.stderr}")
return final_audio_path

View File

@@ -1,85 +0,0 @@
from config import LLM_MODELS # Импортируем словарь моделей
import gradio as gr
from config import GEMINI_API_KEY, IO_API_KEY
class GradioHandlers:
def __init__(self, llm_factory, ConvertMdToPdf, FileHandlers, FasterWhisper, GlueAudio):
# Объект для работы с файлами
self.fh = FileHandlers()
self.ga = GlueAudio()
self.ConvertMdToPdf = ConvertMdToPdf()
self.FasterWhisper = FasterWhisper()
self.llm_factory = llm_factory
def handleRecognizeBtn(
self, audioFiles, model, device, compute_type, beamSize, vadFilter,
minSilenceDurationMs, speechPadMs, temp0, temp1, temp2,
wordTimestamps, noSpeechThreshold, conditionOnPreviousText, filename, outPath
):
try:
glued_audio_path = self.ga.glue(
audio_files=[f.name for f in audioFiles], # Передаем список путей
output_path=outPath,
output_filename=filename
)
except (FileNotFoundError, RuntimeError) as e:
# Если FFmpeg не найден или произошла ошибка, сообщаем пользователю
gr.Warning(str(e))
return "" # Возвращаем пустую строку в текстовое поле
# Передаем путь к склеенному файлу в FasterWhisper
return self.FasterWhisper.recognize(model, device, compute_type, str(glued_audio_path), beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText)
# Функция улучшения текста
def generateByCondition(self, api_key, llm_provider,
llm_model, system_prompt, recognized_text,
llm_temperature, is_pipeline_enabled, trigger,
isSaveFile, filename, filenamePdf, output_path):
try:
# Получаем нужный провайдер через фабрику
provider = self.llm_factory(llm_provider, api_key)
except ValueError as e:
# Если API ключ не предоставлен для нужного провайдера, выводим ошибку
gr.Warning(str(e))
return gr.skip(), gr.skip()
def process():
result, md = provider.generate(llm_model, system_prompt, recognized_text, llm_temperature)
pdf, unicodeText = self.ConvertMdToPdf.convertLatexToText(md)
if isSaveFile:
self.fh.saveFile(filenamePdf, pdf, output_path)
self.fh.saveFile(filename, result, output_path)
return result, unicodeText
if (is_pipeline_enabled and trigger == "change") or (not is_pipeline_enabled and trigger == "click"):
return process()
return gr.skip(), gr.skip()
# НОВАЯ ФУНКЦИЯ для обновления списка моделей
def update_model_dropdown(self, provider):
"""
Вызывается при изменении llmProvider.
Возвращает обновленный компонент Dropdown для моделей.
"""
# Получаем список моделей для выбранного провайдера
models = LLM_MODELS.get(provider, [])
# Выбираем первое значение по умолчанию, если список не пуст
default_value = models[0] if models else None
# Возвращаем обновленный компонент. Используем 'gr' напрямую.
if provider == 'io.net': return gr.update(choices=models, value=default_value), gr.update(label='API key (required for io.net, Gemini)', value=IO_API_KEY, interactive=True)
if provider == 'Gemini': return gr.update(choices=models, value=default_value), gr.update(label='API key (required for Oio.net, Gemini)', value=GEMINI_API_KEY, interactive=True)
if provider == 'gpt4free': return gr.update(choices=models, value=default_value), gr.update(label='API key (required for Oio.net, Gemini)', value="", interactive=True)
# Функция для динамического обновления кнопки
def updateButton(self, isChecked):
if not isChecked:
variant = 'primary'
else:
variant = 'secondary'
return gr.update(interactive=not isChecked, variant=variant)
def updateTextbox(self, isChecked):
return gr.update(visible=isChecked)

View File

@@ -1,113 +1,6 @@
aiofiles==24.1.0
annotated-types==0.7.0
anyio==4.10.0
av==15.1.0
beautifulsoup4==4.13.5
Brotli==1.1.0
bs4==0.0.2
certifi==2025.8.3
cffi==2.0.0
charset-normalizer==3.4.3
click==8.2.1
coloredlogs==15.0.1
colour==0.1.5
cssselect2==0.8.0
ctranslate2==4.6.0
distro==1.9.0
dotenv==0.9.9
exceptiongroup==1.3.0
fastapi==0.116.1
faster-whisper==1.2.0
ffmpeg-python==0.2.0
ffmpy==0.6.1
filelock==3.19.1
flatbuffers==25.2.10
flatlatex==0.15
fonttools==4.59.2
fsspec==2025.9.0
future==1.0.0
gradio==5.44.1
gradio_client==1.12.1
groovy==0.1.2
h11==0.16.0
hf-xet==1.1.9
httpcore==1.0.9
httpx==0.28.1
huggingface-hub==0.34.4
humanfriendly==10.0
idna==3.10
iso639-lang==2.6.3
Jinja2==3.1.6
jiter==0.10.0
joblib==1.5.2
langdetect==1.0.9
littleutils==0.2.4
markdown-it-py==3.0.0
markdown_pdf==1.9
MarkupSafe==3.0.2
mdurl==0.1.2
mpmath==1.3.0
nltk==3.9.1
numpy==2.2.6
onnxruntime==1.22.1
openai==1.106.1
orjson==3.11.3
outdated==0.2.2
packaging==25.0
pandas==2.3.2
pillow==11.3.0
protobuf==6.32.0
pycparser==2.22
pydantic==2.11.7
pydantic_core==2.33.2
pydub==0.25.1
pydyf==0.11.0
Pygments==2.19.2
pylatexenc==2.10
pymultidictionary==1.3.2
PyMuPDF==1.26.4
pyperclip==1.9.0
pyphen==0.17.2
python-dateutil==2.9.0.post0
python-dotenv==1.1.1
python-multipart==0.0.20
pytz==2025.2
PyYAML==6.0.2
regex==2025.9.1
requests==2.32.5
rich==14.1.0
ruff==0.12.12
safehttpx==0.1.6
semantic-version==2.10.0
shellingham==1.5.4
six==1.17.0
sniffio==1.3.1
soupsieve==2.8
starlette==0.47.3
sympy==1.14.0
tinycss2==1.4.0
tinyhtml5==2.0.0
tkmacosx==1.0.5
tokenizers==0.22.0
tomlkit==0.13.3
tqdm==4.67.1
typer==0.17.4
typing-inspection==0.4.1
typing_extensions==4.15.0
tzdata==2025.2
urllib3==2.5.0
uvicorn==0.35.0
webencodings==0.5.1
websockets==15.0.1
zopfli==0.2.3.post1
aiofiles==24.1.0
aiohappyeyeballs==2.6.1
aiohttp==3.13.0
aiosignal==1.4.0
annotated-types==0.7.0
anyio==4.10.0
attrs==25.4.0
audioop-lts==0.2.2
av==15.1.0
beautifulsoup4==4.13.5
Brotli==1.1.0
@@ -131,10 +24,8 @@ filelock==3.19.1
flatbuffers==25.2.10
flatlatex==0.15
fonttools==4.59.2
frozenlist==1.8.0
fsspec==2025.9.0
future==1.0.0
g4f==0.6.3.5
gradio==5.44.1
gradio_client==1.12.1
groovy==0.1.2
@@ -156,8 +47,6 @@ markdown_pdf==1.9
MarkupSafe==3.0.2
mdurl==0.1.2
mpmath==1.3.0
multidict==6.7.0
nest-asyncio==1.6.0
nltk==3.9.1
numpy==2.2.6
onnxruntime==1.22.1
@@ -167,10 +56,8 @@ outdated==0.2.2
packaging==25.0
pandas==2.3.2
pillow==11.3.0
propcache==0.4.0
protobuf==6.32.0
pycparser==2.22
pycryptodome==3.23.0
pydantic==2.11.7
pydantic_core==2.33.2
pydub==0.25.1
@@ -192,7 +79,6 @@ rich==14.1.0
ruff==0.12.12
safehttpx==0.1.6
semantic-version==2.10.0
setuptools==80.9.0
shellingham==1.5.4
six==1.17.0
sniffio==1.3.1
@@ -213,5 +99,4 @@ urllib3==2.5.0
uvicorn==0.35.0
webencodings==0.5.1
websockets==15.0.1
yarl==1.22.0
zopfli==0.2.3.post1

View File

@@ -1,36 +1,29 @@
import re
from pylatexenc.latex2text import LatexNodes2Text
from markdown_pdf import MarkdownPdf
from markdown_pdf import Section
class ConvertMdToPdf:
# Конвертирует md в pdf
def convertLatexToText(self, text:str):
'''
Функция для конвертации LaTeX в текст;
Args:
:param text: текст содержащий LaTeX.
'''
# Обрабатываем только математические выражения
text = re.sub(
r'\$\$(.*?)\$\$|\$(.*?)\$',
self.replace_math,
text,
flags=re.DOTALL
)
pdf = MarkdownPdf(toc_level=0, optimize=True)
pdf.add_section(Section(text))
return pdf, text
def replace_math(self, match):
math_content = match.group(1) or match.group(2) # $$...$$ или $...$
try:
# Преобразуем только математическое выражение
converted = LatexNodes2Text().latex_to_text(math_content)
return converted
except:
return math_content # В случае ошибки оставляем как есть
import re
from pylatexenc.latex2text import LatexNodes2Text
from markdown_pdf import MarkdownPdf
from markdown_pdf import Section
class ConvertMdToPdf:
# Конвертирует md в pdf
def convertLatexToText(self, text:str):
# Обрабатываем только математические выражения
text = re.sub(
r'\$\$(.*?)\$\$|\$(.*?)\$',
self.replace_math,
text,
flags=re.DOTALL
)
pdf = MarkdownPdf(toc_level=0, optimize=True)
pdf.add_section(Section(text))
return pdf, text
def replace_math(self, match):
math_content = match.group(1) or match.group(2) # $$...$$ или $...$
try:
# Преобразуем только математическое выражение
converted = LatexNodes2Text().latex_to_text(math_content)
return converted
except:
return math_content # В случае ошибки оставляем как есть

View File

@@ -1,39 +1,34 @@
from faster_whisper import WhisperModel
class FasterWhisper:
def recognize(self, model, device, compute_type,
audioFile, beamSize, vadFilter,
minSilenceDurationMs, speechPadMs,
temp0, temp1, temp2, wordTimestamps,
noSpeechThreshold, conditionOnPreviousText):
model = WhisperModel(model, device=device, compute_type=compute_type) # Задаем модель
segments, _ = model.transcribe( # Распознаем текст
audioFile,
beam_size=beamSize,
vad_filter=vadFilter,
vad_parameters={
"min_silence_duration_ms": minSilenceDurationMs,
"speech_pad_ms": speechPadMs
},
temperature= [temp0, temp1, temp2],
word_timestamps=wordTimestamps,
no_speech_threshold=noSpeechThreshold,
condition_on_previous_text=conditionOnPreviousText
)
text = ''
for seg in segments:
text += f"[{self.format_timestamp(seg.start)} -> {self.format_timestamp(seg.end)}] {seg.text}" + '\n'
print(f"[{self.format_timestamp(seg.start)} -> {self.format_timestamp(seg.end)}] {seg.text}")
return text
def format_timestamp(self, seconds: float) -> str:
millis = int(seconds * 1000)
hours = millis // (3600 * 1000)
minutes = (millis % (3600 * 1000)) // (60 * 1000)
seconds_int = (millis % (60 * 1000)) // 1000
millis = millis % 1000
return f"{hours:02d}:{minutes:02d}:{seconds_int:02d},{millis:03d}"
from faster_whisper import WhisperModel
class FasterWhisper:
def recognize(self, model, device, compute_type, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText):
model = WhisperModel(model, device=device, compute_type=compute_type) # Задаем модель
segments, _ = model.transcribe( # Распознаем текст
audioFile,
beam_size=beamSize,
vad_filter=vadFilter,
vad_parameters={
"min_silence_duration_ms": minSilenceDurationMs,
"speech_pad_ms": speechPadMs
},
temperature= [temp0, temp1, temp2],
word_timestamps=wordTimestamps,
no_speech_threshold=noSpeechThreshold,
condition_on_previous_text=conditionOnPreviousText
)
text = ''
for seg in segments:
text += f"[{self.format_timestamp(seg.start)} -> {self.format_timestamp(seg.end)}] {seg.text}" + '\n'
return(text)
def format_timestamp(self, seconds: float) -> str:
millis = int(seconds * 1000)
hours = millis // (3600 * 1000)
minutes = (millis % (3600 * 1000)) // (60 * 1000)
seconds_int = (millis % (60 * 1000)) // 1000
millis = millis % 1000
return f"{hours:02d}:{minutes:02d}:{seconds_int:02d},{millis:03d}"

31
services/llm.py Normal file
View File

@@ -0,0 +1,31 @@
import openai
class Llm:
def __init__(self, apiKey:str):
self.client = openai.OpenAI(
api_key=apiKey,
base_url='https://api.intelligence.io.solutions/api/v1/'
)
def generate(self, model:str, systemPrompt:str, userPrompt:str, temp:float):
'''
prompt[user_promtp, system_ptompt]
Function generate text by prompt
'''
# Получаем ответ от нейросети
response = self.client.chat.completions.create(
model=model,
messages=[
{'role': 'system', 'content': systemPrompt},
{'role': 'user', 'content': userPrompt},
],
temperature=temp,
stream=False
)
# Достаем текст
text = str(response.choices[0].message.content)
return text, text

View File

@@ -1,22 +0,0 @@
# services/llm_factory.py
from services.llm_providers.ionet_provider import IoNetProvider
from services.llm_providers.gemini_provider import GeminiProvider
from services.llm_providers.gpt4free_provider import Gpt4FreeProvider
from services.llm_providers.base_provider import BaseLLMProvider
def get_llm_provider(provider_name: str, api_key: str | None) -> BaseLLMProvider:
"""
Фабричная функция для получения экземпляра провайдера LLM.
"""
if provider_name == 'io.net':
if not api_key:
raise ValueError("API ключ обязателен для io.net")
return IoNetProvider(api_key)
elif provider_name == 'Gemini':
if not api_key:
raise ValueError("API ключ обязателен для Gemini")
return GeminiProvider(api_key)
elif provider_name == 'gpt4free':
return Gpt4FreeProvider()
else:
raise ValueError(f"Неизвестный провайдер: {provider_name}")

View File

@@ -1,18 +0,0 @@
from abc import ABC, abstractmethod
class BaseLLMProvider(ABC):
"""
Абстрактный базовый класс для всех провайдеров LLM.
Каждый провайдер должен реализовать метод generate.
"""
def __init__(self, api_key: str | None = None):
self.api_key = api_key
@abstractmethod
def generate(self, model: str, system_prompt: str, user_prompt: str, temp: float):
"""
Основной метод для генерации текста.
Должен возвращать кортеж из двух строк: (чистый_текст, markdown_текст)
"""
pass

View File

@@ -1,74 +0,0 @@
# services/llm_providers/gemini_provider.py
import requests
from .base_provider import BaseLLMProvider
class GeminiProvider(BaseLLMProvider):
"""
Провайдер для Google Gemini, использующий прямые REST API вызовы
через библиотеку requests для надежной работы с SOCKS-прокси.
"""
def __init__(self, api_key: str):
super().__init__(api_key)
self.base_url = "https://generativelanguage.googleapis.com/v1beta/models/"
def generate(self, model: str, system_prompt: str, user_prompt: str, temp: float):
"""
Генерирует текст с помощью модели Gemini, отправляя запрос через прокси.
"""
# 1. Формируем URL для запроса
api_url = f"{self.base_url}{model}:generateContent?key={self.api_key}"
# 2. Задаем настройки прокси из вашего примера
# socks5h:// означает, что DNS-запросы также будут идти через прокси
proxies = {
'http': 'socks5://192.168.1.6:2080',
'https': 'socks5h://192.168.1.6:2080'
}
# 3. Собираем тело запроса (payload) в формате, который ожидает Gemini API
data = {
"system_instruction": {
"parts": {"text": system_prompt}
},
"contents": [{
"parts": [{"text": user_prompt}]
}],
"generationConfig": {
"temperature": temp
}
}
try:
# 4. Отправляем POST-запрос с данными и настройками прокси
response = requests.post(api_url, json=data, proxies=proxies, timeout=90)
# Проверяем, не вернул ли сервер ошибку (например, 4xx или 5xx)
response.raise_for_status()
# 5. Парсим JSON-ответ и извлекаем сгенерированный текст
response_json = response.json()
# Добавим проверку на случай, если контент был заблокирован
if "candidates" not in response_json or not response_json["candidates"]:
block_reason = response_json.get("promptFeedback", {}).get("blockReason", "неизвестная причина")
error_message = f"Контент заблокирован. Причина: {block_reason}"
return error_message, error_message
text = response_json["candidates"][0]["content"]["parts"][0]["text"]
return text, text
except requests.exceptions.ProxyError as e:
error_message = f"Ошибка подключения к прокси. Убедитесь, что Nekobox запущен и слушает порт 2080. Ошибка: {e}"
print(error_message)
return error_message, error_message
except requests.exceptions.RequestException as e:
# Ловим все остальные ошибки requests (таймаут, проблемы с сетью и т.д.)
error_message = f"Произошла ошибка при обращении к API Gemini: {e}"
print(error_message)
return error_message, error_message
except (KeyError, IndexError) as e:
# Ловим ошибки, если структура JSON-ответа неожиданная
error_message = f"Не удалось разобрать ответ от API Gemini. Структура ответа изменилась. Ошибка: {e}"
print(error_message)
return error_message, error_message

View File

@@ -1,28 +0,0 @@
# services/llm_providers/gpt4free_provider.py
from g4f.client import Client
from .base_provider import BaseLLMProvider
class Gpt4FreeProvider(BaseLLMProvider):
# gpt4free не требует API ключа
def __init__(self, api_key: str | None = None):
super().__init__(api_key)
self.client = Client()
def generate(self, model: str, system_prompt: str, user_prompt: str, temp: float):
# temp в g4f может работать не для всех внутренних провайдеров
try:
response = self.client.chat.completions.create(
model=model, # Пример модели, может варьироваться в зависимости от доступности провайдеров
messages=[
{"role": "system", "content": system_prompt},
{"role": "user", "content": user_prompt}
],
temperature=temp
)
text = response.choices[0].message.content
return text, text
except Exception as e:
error_message = f"Ошибка при работе с gpt4free: {e}"
print(error_message)
return error_message, error_message

View File

@@ -1,23 +0,0 @@
import openai
from .base_provider import BaseLLMProvider
class IoNetProvider(BaseLLMProvider):
def __init__(self, api_key: str):
super().__init__(api_key)
self.client = openai.OpenAI(
api_key=self.api_key,
base_url='https://api.intelligence.io.solutions/api/v1/'
)
def generate(self, model: str, system_prompt: str, user_prompt: str, temp: float):
response = self.client.chat.completions.create(
model=model,
messages=[
{'role': 'system', 'content': system_prompt},
{'role': 'user', 'content': user_prompt},
],
temperature=temp,
stream=False
)
text = str(response.choices[0].message.content)
return text, text # Возвращаем как чистый текст, так и Markdown

19
test.py
View File

@@ -1,19 +0,0 @@
from handlers.convertMdToPdf import ConvertMdToPdf
import re
# Создаем экземпляр класса
converter = ConvertMdToPdf()
# Тестовые примеры дробей
test_cases = [
r"$\frac{1}{2}$", # простая дробь
r"$\dfrac{3}{4}$", # дробь с displaystyle
r"$\frac{a}{b} + \frac{c}{d}$", # сложение дробей
r"$$\frac{x^2}{y^3}$$", # блочная дробь
r"$\frac{\partial f}{\partial x}$" # частная производная
]
for latex in test_cases:
result = converter.replace_math(re.search(r'\$\$(.*?)\$\$|\$(.*?)\$', latex))
print(f"Input: {latex}")
print(f"Output: {result}")
print("---")