16 Commits

Author SHA1 Message Date
9bf9d8a55d add Custom provider 2025-12-14 06:31:00 +03:00
1e5105b7d6 fix glueAudio 2025-10-09 04:40:03 +03:00
bc82628147 added API compliance to the provider 2025-10-07 00:32:19 +03:00
bac78a9cee added API compliance to the provider 2025-10-06 22:50:05 +03:00
c2019b57a7 add new LLM provider io.net gemini g4f 2025-10-06 21:17:02 +03:00
Freestyle-Play
6a88c92b9b edit gitcommit 2025-09-25 04:50:30 +03:00
Freestyle-Play
e391a86cb5 add new LLM provider io.net gemini g4f 2025-09-25 04:27:17 +03:00
Voskik
4c90088308 Update .gitignore 2025-09-13 01:50:56 +03:00
swrneko
f2c7dcbeb6 add multiple audiofiles input and glue it 2025-09-11 14:04:21 +03:00
swrneko
6b27149460 add right run style 2025-09-11 13:28:04 +03:00
swrneko
5bf005cd81 remove usless comments from app.py 2025-09-11 13:25:49 +03:00
swrneko
e05e05f829 move pdf handler to handlers from services 2025-09-11 13:25:04 +03:00
swrneko
07485f9a23 move all functions at it's own class in handlers 2025-09-11 13:23:00 +03:00
swrneko
4d4b23177b add some documentations for func in some files 2025-09-10 21:25:10 +03:00
swrneko
26d2d19329 move file logic to it's own class in handlers dir 2025-09-10 21:20:29 +03:00
swrneko
fb7a03b16a move dict with choises to config.py file 2025-09-10 21:02:50 +03:00
21 changed files with 1587 additions and 1175 deletions

8
.gitignore vendored
View File

@@ -1,3 +1,5 @@
.env .env
__pycache__/ __pycache__/
outputs/* outputs/*
venv/

1348
LICENSE

File diff suppressed because it is too large Load Diff

118
README.md
View File

@@ -1,59 +1,59 @@
# FWAL WebUI (Faster Whisper And LLM WebUI) by swrneko # FWAL WebUI (Faster Whisper And LLM WebUI) by swrneko
<div align="center"> <div align="center">
<img src="https://count.getloli.com/get/@swrneko-faster-whisper-llm?theme=rule34"/> <img src="https://count.getloli.com/get/@swrneko-faster-whisper-llm?theme=rule34"/>
</div> </div>
## Screenshots ## Screenshots
<div align="center"> <div align="center">
<div> <div>
<img src="src/1.png" style="object-fit: cover;"/> <img src="src/1.png" style="object-fit: cover;"/>
</div> </div>
<div style="align-items: center;"> <div style="align-items: center;">
<img src="src/2.png" style="width: 49.7%; object-fit: cover;"/> <img src="src/2.png" style="width: 49.7%; object-fit: cover;"/>
<img src="src/3.png" style="width: 49.7%; object-fit: cover;"/> <img src="src/3.png" style="width: 49.7%; object-fit: cover;"/>
</div> </div>
</div> </div>
## Requirements ## Requirements
- python-conda or miniconda; - python-conda or miniconda;
- python 3.10 or above; - python 3.10 or above;
- linux (windows not tested but probably working); - linux (windows not tested but probably working);
Python requirements are listed in `requirements.txt`. Python requirements are listed in `requirements.txt`.
## Installation ## Installation
1. Clone repository: 1. Clone repository:
``` ```
git clone https://github.com/swrneko/faster-whisper-n-ionet-llm.git git clone https://github.com/swrneko/faster-whisper-n-ionet-llm.git
cd faster-whisper-n-ionet-llm cd faster-whisper-n-ionet-llm
``` ```
also (if not insatlled) also (if not insatlled)
- Insatll conda: - Insatll conda:
``` ```
wget https://repo.anaconda.com/miniconda/Miniconda3-latest-Linux-x86_64.sh -O miniconda.sh && bash miniconda.sh wget https://repo.anaconda.com/miniconda/Miniconda3-latest-Linux-x86_64.sh -O miniconda.sh && bash miniconda.sh
``` ```
2. Create virtual env: 2. Create virtual env:
``` ```
conda create -n faster-whisper-n-ionet-llm python=3.10 conda create -n faster-whisper-n-ionet-llm python=3.10
conda activate faster-whisper-n-ionet-llm conda activate faster-whisper-n-ionet-llm
conda install nvidia::cudnn cuda-version=12 conda install nvidia::cudnn cuda-version=12
``` ```
3. Install requirements: 3. Install requirements:
``` ```
pip install -r requirements.txt pip install -r requirements.txt
``` ```
4. Get api key from [io.net](https://ai.io.net/ai/api-keys) and insert into `.env` file (need to create it in root of repository directory). 4. Get api key from [io.net](https://ai.io.net/ai/api-keys) and insert into `.env` file (need to create it in root of repository directory).
It should looks like this: It should looks like this:
``` ```
API_KEY='your_api_key_without_qoutes' API_KEY='your_api_key_without_qoutes'
``` ```
5. Done! Now you can just run it like that: 5. Done! Now you can just run it like that:
```shell ```shell
python app.py python app.py
``` ```

323
app.py
View File

@@ -1,195 +1,128 @@
import gradio as gr import gradio as gr
from pathlib import Path from config import *
from services.llm_factory import get_llm_provider
# Подгрузка сервисов from services.fasterWhisper import FasterWhisper
from services.llm import Llm from handlers.gradioHandler import GradioHandlers
from services.fasterWhisper import FasterWhisper from handlers.fileHandlers import FileHandlers
from services.convertMdToPdf import ConvertMdToPdf from handlers.convertMdToPdf import ConvertMdToPdf
from handlers.glueAudio import GlueAudio
# Загрузка параметров конфигурации
from config import * gh = GradioHandlers(get_llm_provider, ConvertMdToPdf, FileHandlers, FasterWhisper, GlueAudio)
# Функция транскрибации def main():
def generateByCondition(api_key, llm_model, system_prompt, recognized_text, llm_temperature, is_pipeline_enabled, trigger, isSaveFile, filename, filenamePdf): with gr.Blocks() as demo:
llm = Llm(api_key) gr.HTML('<div align=center><h1>Faster Whisper WebUI</h1></div>')
# если чекбокс включен и событие было change → обрабатываем with gr.Row():
if is_pipeline_enabled and trigger == "change": with gr.Tab('Actions'):
result, md = llm.generate(llm_model, system_prompt, recognized_text, llm_temperature) isPipelineEnabledCheckbox = gr.Checkbox(label='is pipeline enabled', value=True, interactive=True)
# Конвертируем текст с латексом в юникод with gr.Row():
pdf, unicodeText = ConvertMdToPdf().convertLatexToText(md) with gr.Accordion(label='Recognization and integration'):
with gr.Column():
if isSaveFile: audioFiles = gr.Files(label='Load audio for transcribe', type="filepath")
savePdf(filenamePdf, pdf) images = gr.Files(label='Upload images', file_types=['image'])
saveFile(filename, result) recognizeBtn = gr.Button('recognize and integrate', variant='primary')
with gr.Accordion(label='Recognized text'):
return result, unicodeText recognizedText = gr.TextArea(label='')
# если чекбокс выключен и событие было click → обрабатываем with gr.Accordion(label='LLM'):
if not is_pipeline_enabled and trigger == "click": with gr.Column():
result, md = llm.generate(llm_model, system_prompt, recognized_text, llm_temperature) refineTextBtn = gr.Button('refine text', variant='secondary', interactive=False)
with gr.Accordion(label='Refined text raw'):
# Конвертируем текст с латексом в юникод refinedText = gr.Textbox(label='', show_copy_button=True)
pdf, unicodeText = ConvertMdToPdf().convertLatexToText(md) with gr.Accordion(label='Refined text md formated'):
refinedTextMD = gr.Markdown(label='')
if isSaveFile:
savePdf(filenamePdf, pdf) with gr.Tab('Settings'):
saveFile(filename, result) with gr.Column():
with gr.Accordion('File settings'):
return result, unicodeText saveFileCheckbox = gr.Checkbox(label='save file', value=True, interactive=True)
filename = gr.Textbox(label='Output filename', value='output.md', interactive=True)
# если нет чекбокса и было событие change filenamePdf = gr.Textbox(label='Output filename for pdf', value='output.pdf', interactive=True)
return gr.skip(), gr.skip()
with gr.Accordion(label='Faster whisper settings'):
with gr.Row():
def savePdf(filename, pdf): with gr.Column():
directory = Path(OUTPUT_PATH) device = gr.Dropdown(label='Device', choices=DEVICES, value=DEVICES[1], interactive=True)
filePath = directory / filename compute_type = gr.Dropdown(label='compute_type', choices=COMPUTE_TYPE, value=COMPUTE_TYPE[0], interactive=True)
filePath.parent.mkdir(parents=True, exist_ok=True) fastWhisperModel = gr.Dropdown(label='Model', choices=FAST_WHISPER_MODELS, value=FAST_WHISPER_MODELS[11], interactive=True)
pdf.save(filePath) beamSize = gr.Number(label='beam_size', value=8, interactive=True)
noSpeechThreshold = gr.Number(label='no_speech_threshold', value=0.5, interactive=True)
# Функция сохранеhния файла vadFilter = gr.Checkbox(label='vad_filter', value=True, interactive=True)
def saveFile(filename, text): wordTimestamps = gr.Checkbox(label='word_timestamps', value=True, interactive=True)
directory = Path(OUTPUT_PATH) conditionOnPreviousText = gr.Checkbox(label='condition_on_previous_text', value=False, interactive=True)
filePath = directory / filename with gr.Column():
filePath.parent.mkdir(parents=True, exist_ok=True) with gr.Accordion(label='Vad parameters'):
filePath.write_text(text, encoding='utf-8') minSilenceDurationMs = gr.Number(label='min_silence_duration_ms', value=300, interactive=True)
speechPadMs = gr.Number(label='speech_pad_ms', value=200, interactive=True)
# ConvertMdToPdf().convert(text) with gr.Accordion(label='Temperature'):
temp0 = gr.Number(label='temp_0', value=0.0, interactive=True)
# Функция для динамического обновления кнопки в зависимости от состояния checkbox temp1 = gr.Number(label='temp_1', value=0.2, interactive=True)
def updateButton(isChecked): temp2 = gr.Number(label='temp_2', value=0.4, interactive=True)
if not isChecked:
variant = 'primary' with gr.Accordion(label='LLM settings'):
else: apiKey = gr.Textbox(label='API key (required for io.net, Gemini)', value=DEFAULT_API_KEY, interactive=True)
variant = 'secondary' with gr.Accordion(label='System prompt'):
systemPrompt = gr.Textbox(label='', value=DEFAULT_SYSTEM_PROMPT, interactive=True)
return gr.update(interactive=not isChecked, variant=variant)
with gr.Row():
llmProvider = gr.Dropdown(label='LLM Provider', choices=LLM_PROVIDERS, value=LLM_PROVIDERS[0], interactive=True)
def updateTextbox(isChecked): llmModel = gr.Dropdown(label='Models', choices=LLM_MODELS[LLM_PROVIDERS[0]], value=LLM_MODELS[LLM_PROVIDERS[0]][1], interactive=True)
return gr.update(visible=isChecked) llmTemperature = gr.Number(label='Temperature', value=0.8, interactive=True)
################################################### # Настройки Custom провайдера
# ____ ___.___ ___. .__ # with gr.Accordion(label='Custom Provider Settings', open=True):
#| | \ | \_ |__ ____ | | ______ _ __# customBaseUrl = gr.Textbox(
#| | / | | __ \_/ __ \| | / _ \ \/ \/ /# label='Base URL',
#| | /| | | \_\ \ ___/| |_( <_> ) / # value='http://127.0.0.1:1234/v1/',
#|______/ |___| |___ /\___ >____/\____/ \/\_/ # interactive=True,
# \/ \/ # visible=False # Скрыто по умолчанию
################################################### )
with gr.Blocks() as demo: # Обработчики событий
gr.HTML(''' isPipelineEnabledCheckbox.change(gh.updateButton, inputs=[isPipelineEnabledCheckbox], outputs=refineTextBtn)
<div align=center> saveFileCheckbox.change(gh.updateTextbox, inputs=saveFileCheckbox, outputs=filename)
<h1> saveFileCheckbox.change(gh.updateTextbox, inputs=saveFileCheckbox, outputs=filenamePdf)
Faster Whisper WebUI
</h1> recognizeBtn.click(
</div> gh.handleRecognizeBtn,
''') inputs=[audioFiles, fastWhisperModel, device, compute_type, beamSize,
vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2,
with gr.Row(): wordTimestamps, noSpeechThreshold, conditionOnPreviousText, gr.State(GLUED_AUDIO_FILENAME), gr.State(OUTPUT_PATH)],
# Вкладка с основным взаимодействием outputs=[recognizedText],
with gr.Tab('Actions'): )
isPipelineEnabledCheckbox = gr.Checkbox(label='is pipeline enabled', value=True, interactive=True)
# --- ИСПРАВЛЕНИЕ: ДОБАВЛЕН customBaseUrl В INPUTS ---
with gr.Row(): recognizedText.change(
with gr.Accordion(label='Recognization and integration'): gh.generateByCondition,
with gr.Column(): inputs=[apiKey, llmProvider, llmModel, systemPrompt, recognizedText, llmTemperature,
audioFile = gr.Audio(label='Load audio for transcribe', type="filepath") isPipelineEnabledCheckbox, gr.State("change"), saveFileCheckbox, filename, filenamePdf, gr.State(OUTPUT_PATH), customBaseUrl],
images = gr.Files(label='Upload images', file_types=['image']) outputs=[refinedText, refinedTextMD]
recognizeBtn = gr.Button('recognize and integrate', variant='primary') )
with gr.Accordion(label='Recognized text'): # Обновление выпадающего списка моделей и поля API ключа
recognizedText = gr.TextArea(label='') llmProvider.change(
gh.update_model_dropdown,
with gr.Accordion(label='LLM'): inputs=llmProvider,
with gr.Column(): outputs=[llmModel, apiKey]
refineTextBtn = gr.Button('refine text', variant='secondary', interactive=False) )
with gr.Accordion(label='Refined text raw'): # Переключение видимости URL для Custom провайдера
refinedText = gr.Textbox(label='', show_copy_button=True) llmProvider.change(
fn=gh.toggle_custom_url,
with gr.Accordion(label='Refined text md formated'): inputs=llmProvider,
refinedTextMD = gr.Markdown(label='') outputs=[customBaseUrl]
)
# Вкладка с настройками
with gr.Tab('Settings'): refineTextBtn.click(
with gr.Column(): gh.generateByCondition,
# Первое поле на всю ширину в акордионе настроек inputs=[apiKey, llmProvider, llmModel, systemPrompt, recognizedText, llmTemperature,
with gr.Accordion('File settings'): isPipelineEnabledCheckbox, gr.State("click"), saveFileCheckbox, filename, filenamePdf, gr.State(OUTPUT_PATH), customBaseUrl],
saveFileCheckbox = gr.Checkbox(label='save file', value=True, interactive=True) outputs=[refinedText, refinedTextMD]
filename = gr.Textbox(label='Output filename', value='output.txt', interactive=True) )
filenamePdf = gr.Textbox(label='Output filename for pdf', value='output.pdf', interactive=True)
demo.launch()
# Акордион настроек faster whisper
with gr.Accordion(label='Faster whisper settings'): if __name__ == '__main__':
with gr.Row(): main()
# Левая колонка в акордионе
with gr.Column():
device = gr.Dropdown(label='Device', choices=["cpu", "cuda"], value="cuda", interactive=True)
compute_type = gr.Dropdown(label='compute_type', choices=["auto", "int8", "float16", "float32"], value="auto", interactive=True)
fastWhisperModel = gr.Dropdown(label='Model', choices=FAST_WHISPER_MODELS, value=FAST_WHISPER_MODELS[11], interactive=True)
beamSize = gr.Number(label='beam_size', value=8, interactive=True)
noSpeechThreshold = gr.Number(label='no_speech_threshold', value=0.5, interactive=True)
vadFilter = gr.Checkbox(label='vad_filter', value=True, interactive=True)
wordTimestamps = gr.Checkbox(label='word_timestamps', value=True, interactive=True)
conditionOnPreviousText = gr.Checkbox(label='condition_on_previous_text', value=False, interactive=True)
# Правая колонка в акордионе
with gr.Column():
with gr.Accordion(label='Vad parameters'):
minSilenceDurationMs = gr.Number(label='min_silence_duration_ms', value=300, interactive=True)
speechPadMs = gr.Number(label='speech_pad_ms', value=200, interactive=True)
with gr.Accordion(label='Temperature'):
temp0 = gr.Number(label='temp_0', value=0.0, interactive=True)
temp1 = gr.Number(label='temp_1', value=0.2, interactive=True)
temp2 = gr.Number(label='temp_2', value=0.4, interactive=True)
# Нижний акордион настроек для api ключа llm
with gr.Accordion(label='ai.io.net api settings'):
apiKey = gr.Textbox(label='API key', value=DEFAULT_API_KEY, interactive=True)
with gr.Accordion(label='System prompt'):
systemPrompt = gr.Textbox(label='', value=DEFAULT_SYSTEM_PROMPT, interactive=True)
with gr.Row():
llmModel = gr.Dropdown(label='models', choices=LLM_MODELS, value=LLM_MODELS[1], interactive=True)
llmTemperature = gr.Number(label='Temperature', value=0.8, interactive=True )
######################################################################
#.____ .__ ___. .__ #
#| | ____ ____ |__| ____ \_ |__ ____ | | ______ _ __#
#| | / _ \ / ___\| |/ ___\ | __ \_/ __ \| | / _ \ \/ \/ /#
#| |__( <_> ) /_/ > \ \___ | \_\ \ ___/| |_( <_> ) / #
#|_______ \____/\___ /|__|\___ > |___ /\___ >____/\____/ \/\_/ #
# \/ /_____/ \/ \/ \/ #
######################################################################
isPipelineEnabledCheckbox.change(updateButton, inputs=[isPipelineEnabledCheckbox], outputs=refineTextBtn)
saveFileCheckbox.change(updateTextbox, inputs=saveFileCheckbox, outputs=filename)
saveFileCheckbox.change(updateTextbox, inputs=saveFileCheckbox, outputs=filenamePdf)
recognizeBtn.click(FasterWhisper().recognize, outputs=[recognizedText], inputs=[fastWhisperModel, device, compute_type, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText])
# Если пайплайн включен то тогда делаем автоматически
# автоматический пайплайн
recognizedText.change(
generateByCondition,
inputs=[apiKey, llmModel, systemPrompt, recognizedText, llmTemperature, isPipelineEnabledCheckbox, gr.State("change"), saveFileCheckbox, filename, filenamePdf],
outputs=[refinedText, refinedTextMD]
)
# ручной запуск по кнопке
refineTextBtn.click(
generateByCondition,
inputs=[apiKey, llmModel, systemPrompt, recognizedText, llmTemperature, isPipelineEnabledCheckbox, gr.State("click"), saveFileCheckbox, filename, filenamePdf],
outputs=[refinedText, refinedTextMD]
)
demo.launch()

141
config.py
View File

@@ -1,48 +1,93 @@
import os import os
from dotenv import load_dotenv from dotenv import load_dotenv
load_dotenv() load_dotenv()
FAST_WHISPER_MODELS = ['tiny', 'base', 'small', 'medium', 'large-v1', 'large-v2', 'large-v3', 'large', 'distil-large-v2', 'distil-large-v3', 'distil-large-v3.5', 'large-v3-turbo', 'turbo'] FAST_WHISPER_MODELS = ['tiny', 'base', 'small', 'medium', 'large-v1', 'large-v2', 'large-v3', 'large', 'distil-large-v2', 'distil-large-v3', 'distil-large-v3.5', 'large-v3-turbo', 'turbo']
LLM_MODELS = ['openai/gpt-oss-120b', 'Qwen/Qwen3-235B-A22B-Thinking-2507', 'deepseek-ai/DeepSeek-R1-0528', 'meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8', 'openai/gpt-oss-20b', 'Intel/Qwen3-Coder-480B-A35B-Instruct-int4-mixed-ar', 'meta-llama/Llama-3.2-90B-Vision-Instruct', 'mistralai/Mistral-Nemo-Instruct-2407', 'Qwen/Qwen2.5-VL-32B-Instruct', 'meta-llama/Llama-3.3-70B-Instruct', 'mistralai/Devstral-Small-2505', 'mistralai/Magistral-Small-2506', 'mistralai/Mistral-Large-Instruct-2411', 'CohereForAI/aya-expanse-32b'] DEVICES = ['cpu', 'cuda']
COMPUTE_TYPE = ['auto', 'int8', 'float16', 'float32']
# Стандартный API ключ # Стандартный API ключ
DEFAULT_API_KEY=os.getenv('API_KEY') IO_API_KEY=os.getenv('IO_API_KEY')
# Задаем выходную директорию GEMINI_API_KEY=os.getenv('GEMINI_API_KEY')
OUTPUT_PATH='outputs' DEFAULT_API_KEY=IO_API_KEY
DEFAULT_SYSTEM_PROMPT='''You are a diligent university student who has recorded a lecture as an audio file and later transcribed it into raw text.
Your task is to rewrite this unstructured transcript into a clear, logically organized, and detailed lecture summary (lecture notes). # Словарь провайдеров и их моделей
LLM_PROVIDERS = ['io.net', 'Gemini', 'gpt4free', 'Custom']
Guidelines: LLM_MODELS = {
1. Structure: 'io.net': [
- Organize the text into a hierarchy of sections and subsections. 'openai/gpt-oss-120b', 'Qwen/Qwen3-235B-A22B-Thinking-2507',
- Use headings, bullet points, or numbering where appropriate. 'deepseek-ai/DeepSeek-R1-0528', 'meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8',
- Present the material in a logical flow (from introduction → main points → details → examples → conclusion). 'openai/gpt-oss-20b', 'Intel/Qwen3-Coder-480B-A35B-Instruct-int4-mixed-ar',
'meta-llama/Llama-3.2-90B-Vision-Instruct', 'mistralai/Mistral-Nemo-Instruct-2407',
2. Clarity & Cohesion: 'Qwen/Qwen2.5-VL-32B-Instruct', 'meta-llama/Llama-3.3-70B-Instruct',
- Remove filler words, repetitions, and irrelevant fragments. 'mistralai/Devstral-Small-2505', 'mistralai/Magistral-Small-2506',
- Rewrite incomplete sentences into full, grammatically correct sentences. 'mistralai/Mistral-Large-Instruct-2411', 'CohereForAI/aya-expanse-32b'
- Ensure smooth transitions between topics, making the summary feel continuous and well-connected. ],
'Gemini': [
3. Depth & Detail: 'gemini-2.5-pro',
- Capture all important concepts, definitions, examples, and explanations from the lecture. 'gemini-2.5-flash',
- Expand shorthand or fragmented thoughts into full, precise explanations. 'gemini-2.5-flash-lite'
- Where appropriate, rephrase or clarify confusing passages for better understanding. ],
'gpt4free': [
4. Accuracy: 'default',
- Preserve the lecturer’s original meaning, intent, and terminology. 'gpt-4',
- Avoid adding personal opinions or new information that was not in the lecture. 'sonar-reasoning',
'command-r-plus',
5. Style: 'llama-3.3-70b',
- Write in a formal, academic tone suitable for study notes. 'hermes-3-llama-3.1-405b'
- Aim for readability: concise sentences, but thorough coverage of concepts. 'qwen-3-235b',
- Use emphasis (e.g., bold or italic text) only when it improves comprehension. 'gpt-4o-mini',
'deepseek-r1',
Final Output: A cohesive, detailed, and well-structured lecture summary, suitable for later studying and revision. 'PollinationsAI:gpt-5-nano'
Use only russian language! ],
USE LATEX IN DOLLAR SIGN ($)! 'Custom': [
EXTRA BIG LENTH OF CONSPECT! 'qwen/qwen3-vl-30b',
''' 'qwen/qwen3-coder-30b',
'openai/gpt-oss-20b',
'qwen3-vl-8b-thinking',
'qwen/qwen3-vl-8b',
],
}
# Задаем выходную директорию
OUTPUT_PATH='outputs'
GLUED_AUDIO_FILENAME='glued.mp3'
DEFAULT_SYSTEM_PROMPT='''You are a diligent university student who has recorded a lecture as an audio file and later transcribed it into raw text.
Your task is to rewrite this unstructured transcript into a clear, logically organized, and detailed lecture summary (lecture notes).
Guidelines:
1. Structure:
- Organize the text into a hierarchy of sections and subsections.
- Use headings, bullet points, or numbering where appropriate.
- Present the material in a logical flow (from introduction → main points → details → examples → conclusion).
2. Clarity & Cohesion:
- Remove filler words, repetitions, and irrelevant fragments.
- Rewrite incomplete sentences into full, grammatically correct sentences.
- Ensure smooth transitions between topics, making the summary feel continuous and well-connected.
3. Depth & Detail:
- Capture all important concepts, definitions, examples, and explanations from the lecture.
- Expand shorthand or fragmented thoughts into full, precise explanations.
- Where appropriate, rephrase or clarify confusing passages for better understanding.
4. Accuracy:
- Preserve the lecturer’s original meaning, intent, and terminology.
- Avoid adding personal opinions or new information that was not in the lecture.
5. Style:
- Write in a formal, academic tone suitable for study notes.
- Aim for readability: concise sentences, but thorough coverage of concepts.
- Use emphasis (e.g., bold or italic text) only when it improves comprehension.
Final Output: A cohesive, detailed, and well-structured lecture summary, suitable for later studying and revision.
Use only russian language!
USE LATEX IN DOLLAR SIGN ($)!
EXTRA BIG LENTH OF CONSPECT!
MAKE AS LONG AS POSIBLE AND AS BE GOOD!
'''

0
handlers/__init__.py Normal file
View File

View File

@@ -1,29 +1,36 @@
import re import re
from pylatexenc.latex2text import LatexNodes2Text from pylatexenc.latex2text import LatexNodes2Text
from markdown_pdf import MarkdownPdf from markdown_pdf import MarkdownPdf
from markdown_pdf import Section from markdown_pdf import Section
class ConvertMdToPdf: class ConvertMdToPdf:
# Конвертирует md в pdf # Конвертирует md в pdf
def convertLatexToText(self, text:str): def convertLatexToText(self, text:str):
# Обрабатываем только математические выражения '''
text = re.sub( Функция для конвертации LaTeX в текст;
r'\$\$(.*?)\$\$|\$(.*?)\$',
self.replace_math, Args:
text, :param text: текст содержащий LaTeX.
flags=re.DOTALL '''
)
pdf = MarkdownPdf(toc_level=0, optimize=True) # Обрабатываем только математические выражения
pdf.add_section(Section(text)) text = re.sub(
return pdf, text r'\$\$(.*?)\$\$|\$(.*?)\$',
self.replace_math,
def replace_math(self, match): text,
math_content = match.group(1) or match.group(2) # $$...$$ или $...$ flags=re.DOTALL
try: )
# Преобразуем только математическое выражение pdf = MarkdownPdf(toc_level=0, optimize=True)
converted = LatexNodes2Text().latex_to_text(math_content) pdf.add_section(Section(text))
return pdf, text
return converted
except: def replace_math(self, match):
return math_content # В случае ошибки оставляем как есть math_content = match.group(1) or match.group(2) # $$...$$ или $...$
try:
# Преобразуем только математическое выражение
converted = LatexNodes2Text().latex_to_text(math_content)
return converted
except:
return math_content # В случае ошибки оставляем как есть

31
handlers/fileHandlers.py Normal file
View File

@@ -0,0 +1,31 @@
from pathlib import Path
# Для аннотации типов
from markdown_pdf import MarkdownPdf
from pydub import AudioSegment
class FileHandlers:
# Функция сохранения файла
def saveFile(self, filename, content, output_path, format='mp3'):
'''
Сохраняет текст, pdf из markdown_pdf или склеенный аудиофайл в файл с указанным названием и директорией.
Args:
:param filename: название файла;
:param content: содержание файла;
:param output_path: выходная диретория файла.
'''
# Создание объекта директории
directory = Path(output_path)
filePath = directory / filename # Добавление пути директории
filePath.parent.mkdir(parents=True, exist_ok=True) # Создание директории если не существует
# Сохранение для разных типов
if type(content) == MarkdownPdf:
return content.save(filePath)
elif type(content) == str:
return filePath.write_text(content, encoding='utf-8')
elif type(content) == AudioSegment:
return content.export(filePath, format=format)

61
handlers/glueAudio.py Normal file
View File

@@ -0,0 +1,61 @@
import subprocess
from pathlib import Path
class GlueAudio():
def glue(self, audio_files: list, output_path: str, output_filename: str) -> Path:
"""
Склеивает аудиофайлы РАЗНЫХ форматов с помощью FFmpeg и filter_complex.
Это универсальный и эффективный по памяти метод.
Args:
audio_files (list): Список путей к исходным аудиофайлам.
output_path (str): Директория для сохранения итогового файла.
output_filename (str): Имя итогового склеенного файла.
Returns:
Path: Путь к созданному склеенному файлу.
"""
output_dir = Path(output_path)
output_dir.mkdir(parents=True, exist_ok=True)
final_audio_path = output_dir / output_filename
if not audio_files:
raise ValueError("Список аудиофайлов для склейки пуст.")
# 1. Формируем часть команды с входными файлами (-i file1 -i file2 ...)
input_args = []
for file_path in audio_files:
input_args.extend(['-i', str(Path(file_path).resolve())])
# 2. Формируем строку для filter_complex
num_files = len(audio_files)
stream_specifiers = "".join([f"[{i}:a]" for i in range(num_files)])
filter_complex_str = f"{stream_specifiers}concat=n={num_files}:v=0:a=1[outa]"
# 3. Собираем полную команду
command = [
'ffmpeg',
*input_args, # Распаковываем список входных файлов
'-filter_complex', filter_complex_str,
'-map', '[outa]',
'-c:a', 'libmp3lame',
'-q:a', '2',
str(final_audio_path),
'-y'
]
try:
# 4. Выполняем команду
print(f"Выполнение команды FFmpeg: {' '.join(command)}")
subprocess.run(command, check=True, capture_output=True, text=True)
print("FFmpeg успешно завершил склейку.")
except FileNotFoundError:
raise FileNotFoundError("FFmpeg не найден. Убедитесь, что он установлен и доступен в системной переменной PATH.")
except subprocess.CalledProcessError as e:
print("Ошибка при выполнении FFmpeg!")
print("Stderr:", e.stderr)
raise RuntimeError(f"Ошибка FFmpeg при склейке файлов: {e.stderr}")
return final_audio_path

87
handlers/gradioHandler.py Normal file
View File

@@ -0,0 +1,87 @@
from config import LLM_MODELS, GEMINI_API_KEY, IO_API_KEY
import gradio as gr
class GradioHandlers:
def __init__(self, llm_factory, ConvertMdToPdf, FileHandlers, FasterWhisper, GlueAudio):
self.fh = FileHandlers()
self.ga = GlueAudio()
self.ConvertMdToPdf = ConvertMdToPdf()
self.FasterWhisper = FasterWhisper()
self.llm_factory = llm_factory
def handleRecognizeBtn(self, audioFiles, model, device, compute_type, beamSize, vadFilter,
minSilenceDurationMs, speechPadMs, temp0, temp1, temp2,
wordTimestamps, noSpeechThreshold, conditionOnPreviousText, filename, outPath):
try:
glued_audio_path = self.ga.glue(
audio_files=[f.name for f in audioFiles],
output_path=outPath,
output_filename=filename
)
except (FileNotFoundError, RuntimeError) as e:
gr.Warning(str(e))
return ""
return self.FasterWhisper.recognize(model, device, compute_type, str(glued_audio_path), beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText)
# Добавил аргумент custom_base_url в конец
def generateByCondition(self, api_key, llm_provider,
llm_model, system_prompt, recognized_text,
llm_temperature, is_pipeline_enabled, trigger,
isSaveFile, filename, filenamePdf, output_path, custom_base_url):
try:
if llm_provider == "Custom":
# Передаем base_url только для Custom
provider = self.llm_factory(llm_provider, api_key, base_url=custom_base_url)
else:
provider = self.llm_factory(llm_provider, api_key)
except ValueError as e:
gr.Warning(str(e))
return gr.skip(), gr.skip()
def process():
# Добавлена обработка ошибок генерации
try:
result, md = provider.generate(llm_model, system_prompt, recognized_text, llm_temperature)
except Exception as e:
raise gr.Error(f"Ошибка генерации LLM: {e}")
pdf, unicodeText = self.ConvertMdToPdf.convertLatexToText(md)
if isSaveFile:
self.fh.saveFile(filenamePdf, pdf, output_path)
self.fh.saveFile(filename, result, output_path)
return result, unicodeText
if (is_pipeline_enabled and trigger == "change") or (not is_pipeline_enabled and trigger == "click"):
return process()
return gr.skip(), gr.skip()
def update_model_dropdown(self, provider):
models = LLM_MODELS.get(provider, [])
default_value = models[0] if models else None
# Обновляем список моделей и настройки поля API Key
if provider == 'io.net':
return gr.update(choices=models, value=default_value), gr.update(label='API key', value=IO_API_KEY, interactive=True, visible=True)
if provider == 'Gemini':
return gr.update(choices=models, value=default_value), gr.update(label='API key', value=GEMINI_API_KEY, interactive=True, visible=True)
if provider == 'gpt4free':
return gr.update(choices=models, value=default_value), gr.update(label='API key (not required)', value="", interactive=False, visible=True)
if provider == "Custom":
return gr.update(choices=models, value=default_value), gr.update(label='API key (optional)', value="", interactive=True, visible=True) # Для Custom ключ может понадобиться
def updateButton(self, isChecked):
variant = 'secondary' if isChecked else 'primary'
return gr.update(interactive=not isChecked, variant=variant)
def toggle_custom_url(self, provider):
"""Показывает поле Base URL только если выбран Custom"""
return gr.update(visible=(provider == 'Custom'))
def update_custom_url(self, base_url):
return None
def updateTextbox(self, isChecked):
return gr.update(visible=isChecked)

View File

@@ -0,0 +1,21 @@
import os
from PIL import Image, ExifTags
import subprocess
import json
from datetime import datetime
class MetadataHandler:
def get_image_timestamp(self, image_path: str) -> datetime | None:
"""Извлекает метку времени из метаданных изображения, если она доступна."""
try:
image = Image.open(image_path)
exif_data = image._getexif()
if exif_data:
for tag, value in exif_data.items():
decoded_tag = ExifTags.TAGS.get(tag, tag)
if decoded_tag == 'DateTimeOriginal':
return value
return None
except Exception as e:
print(f"Error extracting metadata from image: {e}")
return None

7
pandoc.txt Normal file
View File

@@ -0,0 +1,7 @@
pandoc "out.md" -o output1310.pdf \
--pdf-engine=xelatex \
-V geometry:margin=2.5cm \
-V fontsize=12pt \
-V mainfont="Times New Roman" \
-V colorlinks=true \
-V linkcolor=blue\

View File

@@ -1,102 +1,102 @@
aiofiles==24.1.0 aiofiles==24.1.0
annotated-types==0.7.0 annotated-types==0.7.0
anyio==4.10.0 anyio==4.10.0
av==15.1.0 av==15.1.0
beautifulsoup4==4.13.5 beautifulsoup4==4.13.5
Brotli==1.1.0 Brotli==1.1.0
bs4==0.0.2 bs4==0.0.2
certifi==2025.8.3 certifi==2025.8.3
cffi==2.0.0 cffi==2.0.0
charset-normalizer==3.4.3 charset-normalizer==3.4.3
click==8.2.1 click==8.2.1
coloredlogs==15.0.1 coloredlogs==15.0.1
colour==0.1.5 colour==0.1.5
cssselect2==0.8.0 cssselect2==0.8.0
ctranslate2==4.6.0 ctranslate2==4.6.0
distro==1.9.0 distro==1.9.0
dotenv==0.9.9 dotenv==0.9.9
exceptiongroup==1.3.0 exceptiongroup==1.3.0
fastapi==0.116.1 fastapi==0.116.1
faster-whisper==1.2.0 faster-whisper==1.2.0
ffmpeg-python==0.2.0 ffmpeg-python==0.2.0
ffmpy==0.6.1 ffmpy==0.6.1
filelock==3.19.1 filelock==3.19.1
flatbuffers==25.2.10 flatbuffers==25.2.10
flatlatex==0.15 flatlatex==0.15
fonttools==4.59.2 fonttools==4.59.2
fsspec==2025.9.0 fsspec==2025.9.0
future==1.0.0 future==1.0.0
gradio==5.44.1 gradio==5.44.1
gradio_client==1.12.1 gradio_client==1.12.1
groovy==0.1.2 groovy==0.1.2
h11==0.16.0 h11==0.16.0
hf-xet==1.1.9 hf-xet==1.1.9
httpcore==1.0.9 httpcore==1.0.9
httpx==0.28.1 httpx==0.28.1
huggingface-hub==0.34.4 huggingface-hub==0.34.4
humanfriendly==10.0 humanfriendly==10.0
idna==3.10 idna==3.10
iso639-lang==2.6.3 iso639-lang==2.6.3
Jinja2==3.1.6 Jinja2==3.1.6
jiter==0.10.0 jiter==0.10.0
joblib==1.5.2 joblib==1.5.2
langdetect==1.0.9 langdetect==1.0.9
littleutils==0.2.4 littleutils==0.2.4
markdown-it-py==3.0.0 markdown-it-py==3.0.0
markdown_pdf==1.9 markdown_pdf==1.9
MarkupSafe==3.0.2 MarkupSafe==3.0.2
mdurl==0.1.2 mdurl==0.1.2
mpmath==1.3.0 mpmath==1.3.0
nltk==3.9.1 nltk==3.9.1
numpy==2.2.6 numpy==2.2.6
onnxruntime==1.22.1 onnxruntime==1.22.1
openai==1.106.1 openai==1.106.1
orjson==3.11.3 orjson==3.11.3
outdated==0.2.2 outdated==0.2.2
packaging==25.0 packaging==25.0
pandas==2.3.2 pandas==2.3.2
pillow==11.3.0 pillow==11.3.0
protobuf==6.32.0 protobuf==6.32.0
pycparser==2.22 pycparser==2.22
pydantic==2.11.7 pydantic==2.11.7
pydantic_core==2.33.2 pydantic_core==2.33.2
pydub==0.25.1 pydub==0.25.1
pydyf==0.11.0 pydyf==0.11.0
Pygments==2.19.2 Pygments==2.19.2
pylatexenc==2.10 pylatexenc==2.10
pymultidictionary==1.3.2 pymultidictionary==1.3.2
PyMuPDF==1.26.4 PyMuPDF==1.26.4
pyperclip==1.9.0 pyperclip==1.9.0
pyphen==0.17.2 pyphen==0.17.2
python-dateutil==2.9.0.post0 python-dateutil==2.9.0.post0
python-dotenv==1.1.1 python-dotenv==1.1.1
python-multipart==0.0.20 python-multipart==0.0.20
pytz==2025.2 pytz==2025.2
PyYAML==6.0.2 PyYAML==6.0.2
regex==2025.9.1 regex==2025.9.1
requests==2.32.5 requests==2.32.5
rich==14.1.0 rich==14.1.0
ruff==0.12.12 ruff==0.12.12
safehttpx==0.1.6 safehttpx==0.1.6
semantic-version==2.10.0 semantic-version==2.10.0
shellingham==1.5.4 shellingham==1.5.4
six==1.17.0 six==1.17.0
sniffio==1.3.1 sniffio==1.3.1
soupsieve==2.8 soupsieve==2.8
starlette==0.47.3 starlette==0.47.3
sympy==1.14.0 sympy==1.14.0
tinycss2==1.4.0 tinycss2==1.4.0
tinyhtml5==2.0.0 tinyhtml5==2.0.0
tkmacosx==1.0.5 tkmacosx==1.0.5
tokenizers==0.22.0 tokenizers==0.22.0
tomlkit==0.13.3 tomlkit==0.13.3
tqdm==4.67.1 tqdm==4.67.1
typer==0.17.4 typer==0.17.4
typing-inspection==0.4.1 typing-inspection==0.4.1
typing_extensions==4.15.0 typing_extensions==4.15.0
tzdata==2025.2 tzdata==2025.2
urllib3==2.5.0 urllib3==2.5.0
uvicorn==0.35.0 uvicorn==0.35.0
webencodings==0.5.1 webencodings==0.5.1
websockets==15.0.1 websockets==15.0.1
zopfli==0.2.3.post1 zopfli==0.2.3.post1

View File

@@ -1,34 +1,39 @@
from faster_whisper import WhisperModel from faster_whisper import WhisperModel
class FasterWhisper: class FasterWhisper:
def recognize(self, model, device, compute_type, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText): def recognize(self, model, device, compute_type,
model = WhisperModel(model, device=device, compute_type=compute_type) # Задаем модель audioFile, beamSize, vadFilter,
minSilenceDurationMs, speechPadMs,
segments, _ = model.transcribe( # Распознаем текст temp0, temp1, temp2, wordTimestamps,
audioFile, noSpeechThreshold, conditionOnPreviousText):
beam_size=beamSize, model = WhisperModel(model, device=device, compute_type=compute_type) # Задаем модель
vad_filter=vadFilter,
vad_parameters={ segments, _ = model.transcribe( # Распознаем текст
"min_silence_duration_ms": minSilenceDurationMs, audioFile,
"speech_pad_ms": speechPadMs beam_size=beamSize,
}, vad_filter=vadFilter,
temperature= [temp0, temp1, temp2], vad_parameters={
word_timestamps=wordTimestamps, "min_silence_duration_ms": minSilenceDurationMs,
no_speech_threshold=noSpeechThreshold, "speech_pad_ms": speechPadMs
condition_on_previous_text=conditionOnPreviousText },
) temperature= [temp0, temp1, temp2],
word_timestamps=wordTimestamps,
text = '' no_speech_threshold=noSpeechThreshold,
condition_on_previous_text=conditionOnPreviousText
for seg in segments: )
text += f"[{self.format_timestamp(seg.start)} -> {self.format_timestamp(seg.end)}] {seg.text}" + '\n'
text = ''
return(text)
for seg in segments:
def format_timestamp(self, seconds: float) -> str: text += f"[{self.format_timestamp(seg.start)} -> {self.format_timestamp(seg.end)}] {seg.text}" + '\n'
millis = int(seconds * 1000) print(f"[{self.format_timestamp(seg.start)} -> {self.format_timestamp(seg.end)}] {seg.text}")
hours = millis // (3600 * 1000)
minutes = (millis % (3600 * 1000)) // (60 * 1000) return text
seconds_int = (millis % (60 * 1000)) // 1000
millis = millis % 1000 def format_timestamp(self, seconds: float) -> str:
return f"{hours:02d}:{minutes:02d}:{seconds_int:02d},{millis:03d}" millis = int(seconds * 1000)
hours = millis // (3600 * 1000)
minutes = (millis % (3600 * 1000)) // (60 * 1000)
seconds_int = (millis % (60 * 1000)) // 1000
millis = millis % 1000
return f"{hours:02d}:{minutes:02d}:{seconds_int:02d},{millis:03d}"

View File

@@ -1,31 +0,0 @@
import openai
class Llm:
def __init__(self, apiKey:str):
self.client = openai.OpenAI(
api_key=apiKey,
base_url='https://api.intelligence.io.solutions/api/v1/'
)
def generate(self, model:str, systemPrompt:str, userPrompt:str, temp:float):
'''
prompt[user_promtp, system_ptompt]
Function generate text by prompt
'''
# Получаем ответ от нейросети
response = self.client.chat.completions.create(
model=model,
messages=[
{'role': 'system', 'content': systemPrompt},
{'role': 'user', 'content': userPrompt},
],
temperature=temp,
stream=False
)
# Достаем текст
text = str(response.choices[0].message.content)
return text, text

27
services/llm_factory.py Normal file
View File

@@ -0,0 +1,27 @@
# services/llm_factory.py
from services.llm_providers.ionet_provider import IoNetProvider
from services.llm_providers.gemini_provider import GeminiProvider
from services.llm_providers.gpt4free_provider import Gpt4FreeProvider
from services.llm_providers.base_provider import BaseLLMProvider
from services.llm_providers.custom_provider import CustomProvider
def get_llm_provider(provider_name: str, api_key: str | None = None, base_url: str | None = None) -> BaseLLMProvider:
"""
Фабричная функция для получения экземпляра провайдера LLM.
"""
if provider_name == 'io.net':
if not api_key:
raise ValueError("API ключ обязателен для io.net")
return IoNetProvider(api_key)
elif provider_name == 'Gemini':
if not api_key:
raise ValueError("API ключ обязателен для Gemini")
return GeminiProvider(api_key)
elif provider_name == 'gpt4free':
return Gpt4FreeProvider()
elif provider_name == 'Custom':
if not base_url:
raise ValueError("Base URL обязателен для Custom провайдера")
return CustomProvider(api_key, base_url) # base_url будет установлен позже
else:
raise ValueError(f"Неизвестный провайдер: {provider_name}")

View File

@@ -0,0 +1,18 @@
from abc import ABC, abstractmethod
class BaseLLMProvider(ABC):
"""
Абстрактный базовый класс для всех провайдеров LLM.
Каждый провайдер должен реализовать метод generate.
"""
def __init__(self, api_key: str | None = None):
self.api_key = api_key
@abstractmethod
def generate(self, model: str, system_prompt: str, user_prompt: str, temp: float):
"""
Основной метод для генерации текста.
Должен возвращать кортеж из двух строк: (чистый_текст, markdown_текст)
"""
pass

View File

@@ -0,0 +1,44 @@
import requests
from .base_provider import BaseLLMProvider
class CustomProvider(BaseLLMProvider):
def __init__(self, api_key: str | None = None, base_url: str | None = None):
super().__init__(api_key)
self.base_url = base_url or 'http://127.0.0.1:1234/v1/'
# ГАРАНТИРУЕМ наличие слеша в конце URL
if not self.base_url.endswith('/'):
self.base_url += '/'
def generate(self, model: str, system_prompt: str, user_prompt: str, temp: float):
url = f"{self.base_url}chat/completions"
# Некоторые Custom провайдеры (как vLLM или Ollama) могут требовать API Key, даже если он фиктивный
headers = {"Content-Type": "application/json"}
if self.api_key:
headers["Authorization"] = f"Bearer {self.api_key}"
data = {
"model": model,
"messages": [
{'role': 'system', 'content': system_prompt},
{'role': 'user', 'content': user_prompt},
],
"temperature": temp,
"stream": False
}
try:
response = requests.post(url, headers=headers, json=data, timeout=300)
response.raise_for_status()
result = response.json()
# Обработка разных форматов ответа (на всякий случай)
if 'choices' in result and len(result['choices']) > 0:
text = str(result['choices'][0]['message']['content'])
return text, text
else:
return f"Неожиданный ответ от сервера: {result}", str(result)
except requests.exceptions.RequestException as e:
return f"Ошибка при запросе к CustomProvider API: {e}", f"Ошибка: {e}"

View File

@@ -0,0 +1,74 @@
# services/llm_providers/gemini_provider.py
import requests
from .base_provider import BaseLLMProvider
class GeminiProvider(BaseLLMProvider):
"""
Провайдер для Google Gemini, использующий прямые REST API вызовы
через библиотеку requests для надежной работы с SOCKS-прокси.
"""
def __init__(self, api_key: str):
super().__init__(api_key)
self.base_url = "https://generativelanguage.googleapis.com/v1beta/models/"
def generate(self, model: str, system_prompt: str, user_prompt: str, temp: float):
"""
Генерирует текст с помощью модели Gemini, отправляя запрос через прокси.
"""
# 1. Формируем URL для запроса
api_url = f"{self.base_url}{model}:generateContent?key={self.api_key}"
# 2. Задаем настройки прокси из вашего примера
# socks5h:// означает, что DNS-запросы также будут идти через прокси
proxies = {
'http': 'socks5://192.168.1.6:2080',
'https': 'socks5h://192.168.1.6:2080'
}
# 3. Собираем тело запроса (payload) в формате, который ожидает Gemini API
data = {
"system_instruction": {
"parts": {"text": system_prompt}
},
"contents": [{
"parts": [{"text": user_prompt}]
}],
"generationConfig": {
"temperature": temp
}
}
try:
# 4. Отправляем POST-запрос с данными и настройками прокси
response = requests.post(api_url, json=data, proxies=proxies, timeout=400)
# Проверяем, не вернул ли сервер ошибку (например, 4xx или 5xx)
response.raise_for_status()
# 5. Парсим JSON-ответ и извлекаем сгенерированный текст
response_json = response.json()
# Добавим проверку на случай, если контент был заблокирован
if "candidates" not in response_json or not response_json["candidates"]:
block_reason = response_json.get("promptFeedback", {}).get("blockReason", "неизвестная причина")
error_message = f"Контент заблокирован. Причина: {block_reason}"
return error_message, error_message
text = response_json["candidates"][0]["content"]["parts"][0]["text"]
return text, text
except requests.exceptions.ProxyError as e:
error_message = f"Ошибка подключения к прокси. Убедитесь, что Nekobox запущен и слушает порт 2080. Ошибка: {e}"
print(error_message)
return error_message, error_message
except requests.exceptions.RequestException as e:
# Ловим все остальные ошибки requests (таймаут, проблемы с сетью и т.д.)
error_message = f"Произошла ошибка при обращении к API Gemini: {e}"
print(error_message)
return error_message, error_message
except (KeyError, IndexError) as e:
# Ловим ошибки, если структура JSON-ответа неожиданная
error_message = f"Не удалось разобрать ответ от API Gemini. Структура ответа изменилась. Ошибка: {e}"
print(error_message)
return error_message, error_message

View File

@@ -0,0 +1,47 @@
# services/llm_providers/gpt4free_provider.py
from g4f.client import Client
from .base_provider import BaseLLMProvider
class Gpt4FreeProvider(BaseLLMProvider):
"""
Провайдер для работы с моделью GPT через библиотеку gpt4free.
Этот класс реализует интерфейс BaseLLMProvider и предоставляет возможность
взаимодействия с различными LLM через сервис gpt4free, который не требует
API ключа для работы.
"""
# gpt4free не требует API ключа
def __init__(self, api_key: str | None = None):
super().__init__(api_key)
self.client = Client()
def generate(self, model: str, system_prompt: str, user_prompt: str, temp: float):
"""
Генерирует ответ от модели GPT с использованием gpt4free.
Args:
model (str): Название модели для генерации ответа
system_prompt (str): Системное сообщение для контекста
user_prompt (str): Пользовательский запрос
temp (float): Температура генерации ( controls randomness of responses)
Returns:
tuple: Кортеж из двух одинаковых строк - сгенерированного ответа и его копии
"""
# temp в g4f может работать не для всех внутренних провайдеров
try:
response = self.client.chat.completions.create(
model=model, # Пример модели, может варьироваться в зависимости от доступности провайдеров
messages=[
{"role": "system", "content": system_prompt},
{"role": "user", "content": user_prompt}
],
temperature=temp
)
text = response.choices[0].message.content
return text, text
except Exception as e:
error_message = f"Ошибка при работе с gpt4free: {e}"
print(error_message)
return error_message, error_message

View File

@@ -0,0 +1,34 @@
import requests
from .base_provider import BaseLLMProvider
class IoNetProvider(BaseLLMProvider):
def __init__(self, api_key: str):
super().__init__(api_key)
self.base_url = 'https://api.intelligence.io.solutions/api/v1'
def generate(self, model: str, system_prompt: str, user_prompt: str, temp: float):
url = f"{self.base_url}/chat/completions"
headers = {
"Content-Type": "application/json",
"Authorization": f"Bearer {self.api_key}"
}
data = {
"model": model,
"messages": [
{'role': 'system', 'content': system_prompt},
{'role': 'user', 'content': user_prompt},
],
"temperature": temp
}
try:
response = requests.post(url, headers=headers, json=data)
response.raise_for_status()
result = response.json()
text = str(result['choices'][0]['message']['content'])
return text, text # Возвращаем как чистый текст, так и Markdown
except requests.exceptions.RequestException as e:
raise Exception(f"Ошибка при запросе к IO.net API: {e}")