1 Commits
voksik ... main

Author SHA1 Message Date
swrneko
d231707572 change prompt 2025-12-14 18:37:53 +03:00
13 changed files with 152 additions and 323 deletions

91
app.py
View File

@@ -1,9 +1,17 @@
import gradio as gr import gradio as gr
# Загрузка параметров конфигурации
from config import * from config import *
# Подгрузка сервисов
from services.llm_factory import get_llm_provider from services.llm_factory import get_llm_provider
from services.fasterWhisper import FasterWhisper
# Загрузка доп. модулей
from handlers.gradioHandler import GradioHandlers from handlers.gradioHandler import GradioHandlers
from handlers.fileHandlers import FileHandlers from handlers.fileHandlers import FileHandlers
from services.fasterWhisper import FasterWhisper
from handlers.convertMdToPdf import ConvertMdToPdf from handlers.convertMdToPdf import ConvertMdToPdf
from handlers.glueAudio import GlueAudio from handlers.glueAudio import GlueAudio
@@ -11,9 +19,16 @@ gh = GradioHandlers(get_llm_provider, ConvertMdToPdf, FileHandlers, FasterWhispe
def main(): def main():
with gr.Blocks() as demo: with gr.Blocks() as demo:
gr.HTML('<div align=center><h1>Faster Whisper WebUI</h1></div>') gr.HTML('''
<div align=center>
<h1>
Faster Whisper WebUI
</h1>
</div>
''')
with gr.Row(): with gr.Row():
# Вкладка с основным взаимодействием
with gr.Tab('Actions'): with gr.Tab('Actions'):
isPipelineEnabledCheckbox = gr.Checkbox(label='is pipeline enabled', value=True, interactive=True) isPipelineEnabledCheckbox = gr.Checkbox(label='is pipeline enabled', value=True, interactive=True)
@@ -23,102 +38,104 @@ def main():
audioFiles = gr.Files(label='Load audio for transcribe', type="filepath") audioFiles = gr.Files(label='Load audio for transcribe', type="filepath")
images = gr.Files(label='Upload images', file_types=['image']) images = gr.Files(label='Upload images', file_types=['image'])
recognizeBtn = gr.Button('recognize and integrate', variant='primary') recognizeBtn = gr.Button('recognize and integrate', variant='primary')
with gr.Accordion(label='Recognized text'): with gr.Accordion(label='Recognized text'):
recognizedText = gr.TextArea(label='') recognizedText = gr.TextArea(label='')
with gr.Accordion(label='LLM'): with gr.Accordion(label='LLM'):
with gr.Column(): with gr.Column():
refineTextBtn = gr.Button('refine text', variant='secondary', interactive=False) refineTextBtn = gr.Button('refine text', variant='secondary', interactive=False)
with gr.Accordion(label='Refined text raw'): with gr.Accordion(label='Refined text raw'):
refinedText = gr.Textbox(label='', show_copy_button=True) refinedText = gr.Textbox(label='', show_copy_button=True)
with gr.Accordion(label='Refined text md formated'): with gr.Accordion(label='Refined text md formated'):
refinedTextMD = gr.Markdown(label='') refinedTextMD = gr.Markdown(label='')
# Вкладка с настройками
with gr.Tab('Settings'): with gr.Tab('Settings'):
with gr.Column(): with gr.Column():
# Первое поле на всю ширину в акордионе настроек
with gr.Accordion('File settings'): with gr.Accordion('File settings'):
saveFileCheckbox = gr.Checkbox(label='save file', value=True, interactive=True) saveFileCheckbox = gr.Checkbox(label='save file', value=True, interactive=True)
filename = gr.Textbox(label='Output filename', value='output.md', interactive=True) filename = gr.Textbox(label='Output filename', value='output.txt', interactive=True)
filenamePdf = gr.Textbox(label='Output filename for pdf', value='output.pdf', interactive=True) filenamePdf = gr.Textbox(label='Output filename for pdf', value='output.pdf', interactive=True)
# Акордион настроек faster whisper
with gr.Accordion(label='Faster whisper settings'): with gr.Accordion(label='Faster whisper settings'):
with gr.Row(): with gr.Row():
# Левая колонка в акордионе
with gr.Column(): with gr.Column():
device = gr.Dropdown(label='Device', choices=DEVICES, value=DEVICES[1], interactive=True) device = gr.Dropdown(label='Device', choices=DEVICES, value=DEVICES[1], interactive=True)
compute_type = gr.Dropdown(label='compute_type', choices=COMPUTE_TYPE, value=COMPUTE_TYPE[0], interactive=True) compute_type = gr.Dropdown(label='compute_type', choices=COMPUTE_TYPE, value=COMPUTE_TYPE[0], interactive=True)
fastWhisperModel = gr.Dropdown(label='Model', choices=FAST_WHISPER_MODELS, value=FAST_WHISPER_MODELS[11], interactive=True) fastWhisperModel = gr.Dropdown(label='Model', choices=FAST_WHISPER_MODELS, value=FAST_WHISPER_MODELS[11], interactive=True)
beamSize = gr.Number(label='beam_size', value=8, interactive=True) beamSize = gr.Number(label='beam_size', value=8, interactive=True)
noSpeechThreshold = gr.Number(label='no_speech_threshold', value=0.5, interactive=True) noSpeechThreshold = gr.Number(label='no_speech_threshold', value=0.5, interactive=True)
vadFilter = gr.Checkbox(label='vad_filter', value=True, interactive=True) vadFilter = gr.Checkbox(label='vad_filter', value=True, interactive=True)
wordTimestamps = gr.Checkbox(label='word_timestamps', value=True, interactive=True) wordTimestamps = gr.Checkbox(label='word_timestamps', value=True, interactive=True)
conditionOnPreviousText = gr.Checkbox(label='condition_on_previous_text', value=False, interactive=True) conditionOnPreviousText = gr.Checkbox(label='condition_on_previous_text', value=False, interactive=True)
# Правая колонка в акордионе
with gr.Column(): with gr.Column():
with gr.Accordion(label='Vad parameters'): with gr.Accordion(label='Vad parameters'):
minSilenceDurationMs = gr.Number(label='min_silence_duration_ms', value=300, interactive=True) minSilenceDurationMs = gr.Number(label='min_silence_duration_ms', value=300, interactive=True)
speechPadMs = gr.Number(label='speech_pad_ms', value=200, interactive=True) speechPadMs = gr.Number(label='speech_pad_ms', value=200, interactive=True)
with gr.Accordion(label='Temperature'): with gr.Accordion(label='Temperature'):
temp0 = gr.Number(label='temp_0', value=0.0, interactive=True) temp0 = gr.Number(label='temp_0', value=0.0, interactive=True)
temp1 = gr.Number(label='temp_1', value=0.2, interactive=True) temp1 = gr.Number(label='temp_1', value=0.2, interactive=True)
temp2 = gr.Number(label='temp_2', value=0.4, interactive=True) temp2 = gr.Number(label='temp_2', value=0.4, interactive=True)
# Нижний акордион настроек для api ключа llm
with gr.Accordion(label='LLM settings'): with gr.Accordion(label='LLM settings'):
apiKey = gr.Textbox(label='API key (required for io.net, Gemini)', value=DEFAULT_API_KEY, interactive=True) apiKey = gr.Textbox(label='API key (required for io.net, Gemini)', value=DEFAULT_API_KEY, interactive=True)
with gr.Accordion(label='System prompt'): with gr.Accordion(label='System prompt'):
systemPrompt = gr.Textbox(label='', value=DEFAULT_SYSTEM_PROMPT, interactive=True) systemPrompt = gr.Textbox(label='', value=DEFAULT_SYSTEM_PROMPT, interactive=True)
with gr.Row(): with gr.Row():
llmProvider = gr.Dropdown(label='LLM Provider', choices=LLM_PROVIDERS, value=LLM_PROVIDERS[0], interactive=True) # ВЫБОР ПРОВАЙДЕРА
llmModel = gr.Dropdown(label='Models', choices=LLM_MODELS[LLM_PROVIDERS[0]], value=LLM_MODELS[LLM_PROVIDERS[0]][1], interactive=True) llmProvider = gr.Dropdown(
label='LLM Provider',
choices=LLM_PROVIDERS,
value=LLM_PROVIDERS[0],
interactive=True
)
# СПИСОК МОДЕЛЕЙ (теперь зависит от провайдера)
llmModel = gr.Dropdown(
label='Models',
choices=LLM_MODELS[LLM_PROVIDERS[0]], # Модели для провайдера по умолчанию
value=LLM_MODELS[LLM_PROVIDERS[0]][1],
interactive=True
)
llmTemperature = gr.Number(label='Temperature', value=0.8, interactive=True) llmTemperature = gr.Number(label='Temperature', value=0.8, interactive=True)
# Настройки Custom провайдера
with gr.Accordion(label='Custom Provider Settings', open=True):
customBaseUrl = gr.Textbox(
label='Base URL',
value='http://127.0.0.1:1234/v1/',
interactive=True,
visible=False # Скрыто по умолчанию
)
# Обработчики событий
isPipelineEnabledCheckbox.change(gh.updateButton, inputs=[isPipelineEnabledCheckbox], outputs=refineTextBtn) isPipelineEnabledCheckbox.change(gh.updateButton, inputs=[isPipelineEnabledCheckbox], outputs=refineTextBtn)
saveFileCheckbox.change(gh.updateTextbox, inputs=saveFileCheckbox, outputs=filename) saveFileCheckbox.change(gh.updateTextbox, inputs=saveFileCheckbox, outputs=filename)
saveFileCheckbox.change(gh.updateTextbox, inputs=saveFileCheckbox, outputs=filenamePdf) saveFileCheckbox.change(gh.updateTextbox, inputs=saveFileCheckbox, outputs=filenamePdf)
recognizeBtn.click( recognizeBtn.click(gh.handleRecognizeBtn, outputs=[recognizedText], inputs=[audioFiles, fastWhisperModel, device, compute_type, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText, gr.State(GLUED_AUDIO_FILENAME), gr.State(OUTPUT_PATH)])
gh.handleRecognizeBtn,
inputs=[audioFiles, fastWhisperModel, device, compute_type, beamSize,
vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2,
wordTimestamps, noSpeechThreshold, conditionOnPreviousText, gr.State(GLUED_AUDIO_FILENAME), gr.State(OUTPUT_PATH)],
outputs=[recognizedText],
)
# --- ИСПРАВЛЕНИЕ: ДОБАВЛЕН customBaseUrl В INPUTS --- # Если пайплайн включен то тогда делаем автоматически
# автоматический пайплайн
recognizedText.change( recognizedText.change(
gh.generateByCondition, gh.generateByCondition,
inputs=[apiKey, llmProvider, llmModel, systemPrompt, recognizedText, llmTemperature, inputs=[apiKey, llmProvider, llmModel, systemPrompt, recognizedText, llmTemperature, isPipelineEnabledCheckbox, gr.State("change"), saveFileCheckbox, filename, filenamePdf, gr.State(OUTPUT_PATH)],
isPipelineEnabledCheckbox, gr.State("change"), saveFileCheckbox, filename, filenamePdf, gr.State(OUTPUT_PATH), customBaseUrl],
outputs=[refinedText, refinedTextMD] outputs=[refinedText, refinedTextMD]
) )
# Обновление выпадающего списка моделей и поля API ключа
# ручной запуск по кнопке
llmProvider.change( llmProvider.change(
gh.update_model_dropdown, gh.update_model_dropdown,
inputs=llmProvider, inputs=llmProvider,
outputs=[llmModel, apiKey] outputs=llmModel
) )
# Переключение видимости URL для Custom провайдера
llmProvider.change(
fn=gh.toggle_custom_url,
inputs=llmProvider,
outputs=[customBaseUrl]
)
refineTextBtn.click( refineTextBtn.click(
gh.generateByCondition, gh.generateByCondition,
inputs=[apiKey, llmProvider, llmModel, systemPrompt, recognizedText, llmTemperature, inputs=[apiKey, llmProvider, llmModel, systemPrompt, recognizedText, llmTemperature, isPipelineEnabledCheckbox, gr.State("click"), saveFileCheckbox, filename, filenamePdf, gr.State(OUTPUT_PATH)],
isPipelineEnabledCheckbox, gr.State("click"), saveFileCheckbox, filename, filenamePdf, gr.State(OUTPUT_PATH), customBaseUrl],
outputs=[refinedText, refinedTextMD] outputs=[refinedText, refinedTextMD]
) )

View File

@@ -8,13 +8,10 @@ DEVICES = ['cpu', 'cuda']
COMPUTE_TYPE = ['auto', 'int8', 'float16', 'float32'] COMPUTE_TYPE = ['auto', 'int8', 'float16', 'float32']
# Стандартный API ключ # Стандартный API ключ
IO_API_KEY=os.getenv('IO_API_KEY') DEFAULT_API_KEY=os.getenv('API_KEY')
GEMINI_API_KEY=os.getenv('GEMINI_API_KEY')
DEFAULT_API_KEY=IO_API_KEY
# Словарь провайдеров и их моделей # Словарь провайдеров и их моделей
LLM_PROVIDERS = ['io.net', 'Gemini', 'gpt4free', 'Custom'] LLM_PROVIDERS = ['io.net', 'Gemini', 'gpt4free']
LLM_MODELS = { LLM_MODELS = {
'io.net': [ 'io.net': [
'openai/gpt-oss-120b', 'Qwen/Qwen3-235B-A22B-Thinking-2507', 'openai/gpt-oss-120b', 'Qwen/Qwen3-235B-A22B-Thinking-2507',
@@ -30,7 +27,7 @@ LLM_MODELS = {
'gemini-2.5-flash', 'gemini-2.5-flash',
'gemini-2.5-flash-lite' 'gemini-2.5-flash-lite'
], ],
'gpt4free': [ 'gpt4free': [ # Модели могут меняться, проверьте документацию g4f
'default', 'default',
'gpt-4', 'gpt-4',
'sonar-reasoning', 'sonar-reasoning',
@@ -41,14 +38,7 @@ LLM_MODELS = {
'gpt-4o-mini', 'gpt-4o-mini',
'deepseek-r1', 'deepseek-r1',
'PollinationsAI:gpt-5-nano' 'PollinationsAI:gpt-5-nano'
], ]
'Custom': [
'qwen/qwen3-vl-30b',
'qwen/qwen3-coder-30b',
'openai/gpt-oss-20b',
'qwen3-vl-8b-thinking',
'qwen/qwen3-vl-8b',
],
} }
# Задаем выходную директорию # Задаем выходную директорию
@@ -56,38 +46,42 @@ OUTPUT_PATH='outputs'
GLUED_AUDIO_FILENAME='glued.mp3' GLUED_AUDIO_FILENAME='glued.mp3'
DEFAULT_SYSTEM_PROMPT='''You are a diligent university student who has recorded a lecture as an audio file and later transcribed it into raw text. DEFAULT_SYSTEM_PROMPT='''
Your task is to rewrite this unstructured transcript into a clear, logically organized, and detailed lecture summary (lecture notes). You are a smart university student creating easy-to-understand study notes summary of lesson for a classmate who is a beginner. Your source is a raw text/audio transcript.
Guidelines: GOAL: rewrite the information into a clear, structured summary in RUSSIAN.
1. Structure:
- Organize the text into a hierarchy of sections and subsections.
- Use headings, bullet points, or numbering where appropriate.
- Present the material in a logical flow (from introduction → main points → details → examples → conclusion).
2. Clarity & Cohesion: KEY RULES FOR CONTENT:
- Remove filler words, repetitions, and irrelevant fragments. 1. **Logical Structure:** Use Markdown headers (#, ##), bullet points, and short paragraphs.
- Rewrite incomplete sentences into full, grammatically correct sentences. 2. **No "Water":** Remove filler words. Keep only practical information.
- Ensure smooth transitions between topics, making the summary feel continuous and well-connected. 3. **Student Tone:** Write naturally, as if sharing notes with a friend. Avoid robotic phrases like "It is important to note".
3. Depth & Detail: KEY RULES FOR LATEX (CRITICAL FOR PYLATEXENC):
- Capture all important concepts, definitions, examples, and explanations from the lecture. 1. **Math Mode:** ANY variable (like t, L, C), number in a formula, or equation MUST be wrapped in dollar signs `$`.
- Expand shorthand or fragmented thoughts into full, precise explanations. * BAD: i(t) = i_pr + i_sv
- Where appropriate, rephrase or clarify confusing passages for better understanding. * GOOD: $i(t) = i_{pr} + i_{sv}$
2. **Subscripts:** Always use curly braces `{}` for subscripts longer than one character.
* BAD: $i_pr$
* GOOD: $i_{pr}$ (or $i_{пр}$ if using cyrillic)
3. **Symbols:** Use standard LaTeX commands for symbols.
* Arrow: use `\to` (e.g., $t \to \infty$).
* Infinity: use `\infty`.
* Multiplication: use `\cdot` or just space.
4. **Consistency:** Never leave a mathematical symbol as plain text. If you mention "current i", write "ток $i$".
4. Accuracy: EXAMPLE OF DESIRED OUTPUT FORMAT:
- Preserve the lecturer’s original meaning, intent, and terminology. # Тема лекции
- Avoid adding personal opinions or new information that was not in the lecture. ## Основные понятия
* **Переходный процесс** — это когда цепь перестраивается с одного режима на другой (например, щелкнули выключателем).
* Математически это описывается дифференциальными уравнениями. Порядок уравнения = количеству реактивных элементов ($L$ и $C$).
5. Style: ## Классический метод
- Write in a formal, academic tone suitable for study notes. Решение ищется в виде суммы двух частей:
- Aim for readability: concise sentences, but thorough coverage of concepts. $$i(t) = i_{pr} + i_{sv}$$
- Use emphasis (e.g., bold or italic text) only when it improves comprehension.
Final Output: A cohesive, detailed, and well-structured lecture summary, suitable for later studying and revision. 1. **Принужденная составляющая** ($i_{pr}$) — это режим, который установится в будущем, когда все успокоится ($t \to \infty$).
Use only russian language! 2. **Свободная составляющая** ($i_{sv}$) — это то, что происходит "само по себе" из-за энергии, запасенной в $L$ и $C$.
USE LATEX IN DOLLAR SIGN ($)!
EXTRA BIG LENTH OF CONSPECT! ***
MAKE AS LONG AS POSIBLE AND AS BE GOOD! STRICTLY FOLLOW THESE FORMATTING RULES. OUTPUT IN RUSSIAN.
''' '''

View File

@@ -1,61 +1,11 @@
import subprocess from pydub import AudioSegment
from pathlib import Path
class GlueAudio(): class GlueAudio():
def glue(self, audio_files: list, output_path: str, output_filename: str) -> Path: def glue(self, audioFiles):
""" glued = AudioSegment.empty()
Склеивает аудиофайлы РАЗНЫХ форматов с помощью FFmpeg и filter_complex.
Это универсальный и эффективный по памяти метод.
Args: for audioFile in audioFiles:
audio_files (list): Список путей к исходным аудиофайлам. audio = AudioSegment.from_file(audioFile)
output_path (str): Директория для сохранения итогового файла. glued += audio
output_filename (str): Имя итогового склеенного файла.
Returns: return glued
Path: Путь к созданному склеенному файлу.
"""
output_dir = Path(output_path)
output_dir.mkdir(parents=True, exist_ok=True)
final_audio_path = output_dir / output_filename
if not audio_files:
raise ValueError("Список аудиофайлов для склейки пуст.")
# 1. Формируем часть команды с входными файлами (-i file1 -i file2 ...)
input_args = []
for file_path in audio_files:
input_args.extend(['-i', str(Path(file_path).resolve())])
# 2. Формируем строку для filter_complex
num_files = len(audio_files)
stream_specifiers = "".join([f"[{i}:a]" for i in range(num_files)])
filter_complex_str = f"{stream_specifiers}concat=n={num_files}:v=0:a=1[outa]"
# 3. Собираем полную команду
command = [
'ffmpeg',
*input_args, # Распаковываем список входных файлов
'-filter_complex', filter_complex_str,
'-map', '[outa]',
'-c:a', 'libmp3lame',
'-q:a', '2',
str(final_audio_path),
'-y'
]
try:
# 4. Выполняем команду
print(f"Выполнение команды FFmpeg: {' '.join(command)}")
subprocess.run(command, check=True, capture_output=True, text=True)
print("FFmpeg успешно завершил склейку.")
except FileNotFoundError:
raise FileNotFoundError("FFmpeg не найден. Убедитесь, что он установлен и доступен в системной переменной PATH.")
except subprocess.CalledProcessError as e:
print("Ошибка при выполнении FFmpeg!")
print("Stderr:", e.stderr)
raise RuntimeError(f"Ошибка FFmpeg при склейке файлов: {e.stderr}")
return final_audio_path

View File

@@ -1,52 +1,34 @@
from config import LLM_MODELS, GEMINI_API_KEY, IO_API_KEY from config import LLM_MODELS # Импортируем словарь моделей
import gradio as gr import gradio as gr
class GradioHandlers: class GradioHandlers:
def __init__(self, llm_factory, ConvertMdToPdf, FileHandlers, FasterWhisper, GlueAudio): def __init__(self, llm_factory, ConvertMdToPdf, FileHandlers, FasterWhisper, GlueAudio):
# Объект для работы с файлами
self.fh = FileHandlers() self.fh = FileHandlers()
self.ga = GlueAudio() self.ga = GlueAudio()
self.ConvertMdToPdf = ConvertMdToPdf() self.ConvertMdToPdf = ConvertMdToPdf()
self.FasterWhisper = FasterWhisper() self.FasterWhisper = FasterWhisper()
self.llm_factory = llm_factory self.llm_factory = llm_factory # Сохраняем фабрику
def handleRecognizeBtn(self, audioFiles, model, device, compute_type, beamSize, vadFilter, def handleRecognizeBtn(self, audioFiles, model, device, compute_type, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText, filename, outPath):
minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, audioFile = self.ga.glue(audioFiles)
wordTimestamps, noSpeechThreshold, conditionOnPreviousText, filename, outPath): file = self.fh.saveFile(filename, audioFile, outPath)
return self.FasterWhisper.recognize(model, device, compute_type, file, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText)
# Функция улучшения текста
def generateByCondition(self, api_key, llm_provider, llm_model, system_prompt, recognized_text, llm_temperature, is_pipeline_enabled, trigger, isSaveFile, filename, filenamePdf, output_path):
try: try:
glued_audio_path = self.ga.glue( # Получаем нужный провайдер через фабрику
audio_files=[f.name for f in audioFiles],
output_path=outPath,
output_filename=filename
)
except (FileNotFoundError, RuntimeError) as e:
gr.Warning(str(e))
return ""
return self.FasterWhisper.recognize(model, device, compute_type, str(glued_audio_path), beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText)
# Добавил аргумент custom_base_url в конец
def generateByCondition(self, api_key, llm_provider,
llm_model, system_prompt, recognized_text,
llm_temperature, is_pipeline_enabled, trigger,
isSaveFile, filename, filenamePdf, output_path, custom_base_url):
try:
if llm_provider == "Custom":
# Передаем base_url только для Custom
provider = self.llm_factory(llm_provider, api_key, base_url=custom_base_url)
else:
provider = self.llm_factory(llm_provider, api_key) provider = self.llm_factory(llm_provider, api_key)
except ValueError as e: except ValueError as e:
gr.Warning(str(e)) # Если API ключ не предоставлен для нужного провайдера, выводим ошибку
return gr.skip(), gr.skip() self.gr.Warning(str(e))
return self.gr.skip(), self.gr.skip()
def process(): def process():
# Добавлена обработка ошибок генерации
try:
result, md = provider.generate(llm_model, system_prompt, recognized_text, llm_temperature) result, md = provider.generate(llm_model, system_prompt, recognized_text, llm_temperature)
except Exception as e:
raise gr.Error(f"Ошибка генерации LLM: {e}")
pdf, unicodeText = self.ConvertMdToPdf.convertLatexToText(md) pdf, unicodeText = self.ConvertMdToPdf.convertLatexToText(md)
if isSaveFile: if isSaveFile:
self.fh.saveFile(filenamePdf, pdf, output_path) self.fh.saveFile(filenamePdf, pdf, output_path)
@@ -58,30 +40,28 @@ class GradioHandlers:
return gr.skip(), gr.skip() return gr.skip(), gr.skip()
# НОВАЯ ФУНКЦИЯ для обновления списка моделей
def update_model_dropdown(self, provider): def update_model_dropdown(self, provider):
"""
Вызывается при изменении llmProvider.
Возвращает обновленный компонент Dropdown для моделей.
"""
# Получаем список моделей для выбранного провайдера
models = LLM_MODELS.get(provider, []) models = LLM_MODELS.get(provider, [])
# Выбираем первое значение по умолчанию, если список не пуст
default_value = models[0] if models else None default_value = models[0] if models else None
# Обновляем список моделей и настройки поля API Key # Возвращаем обновленный компонент. Используем 'gr' напрямую.
if provider == 'io.net': return gr.update(choices=models, value=default_value)
return gr.update(choices=models, value=default_value), gr.update(label='API key', value=IO_API_KEY, interactive=True, visible=True)
if provider == 'Gemini':
return gr.update(choices=models, value=default_value), gr.update(label='API key', value=GEMINI_API_KEY, interactive=True, visible=True)
if provider == 'gpt4free':
return gr.update(choices=models, value=default_value), gr.update(label='API key (not required)', value="", interactive=False, visible=True)
if provider == "Custom":
return gr.update(choices=models, value=default_value), gr.update(label='API key (optional)', value="", interactive=True, visible=True) # Для Custom ключ может понадобиться
# Функция для динамического обновления кнопки
def updateButton(self, isChecked): def updateButton(self, isChecked):
variant = 'secondary' if isChecked else 'primary' if not isChecked:
variant = 'primary'
else:
variant = 'secondary'
return gr.update(interactive=not isChecked, variant=variant) return gr.update(interactive=not isChecked, variant=variant)
def toggle_custom_url(self, provider):
"""Показывает поле Base URL только если выбран Custom"""
return gr.update(visible=(provider == 'Custom'))
def update_custom_url(self, base_url):
return None
def updateTextbox(self, isChecked): def updateTextbox(self, isChecked):
return gr.update(visible=isChecked) return gr.update(visible=isChecked)

View File

@@ -1,21 +0,0 @@
import os
from PIL import Image, ExifTags
import subprocess
import json
from datetime import datetime
class MetadataHandler:
def get_image_timestamp(self, image_path: str) -> datetime | None:
"""Извлекает метку времени из метаданных изображения, если она доступна."""
try:
image = Image.open(image_path)
exif_data = image._getexif()
if exif_data:
for tag, value in exif_data.items():
decoded_tag = ExifTags.TAGS.get(tag, tag)
if decoded_tag == 'DateTimeOriginal':
return value
return None
except Exception as e:
print(f"Error extracting metadata from image: {e}")
return None

View File

@@ -1,7 +0,0 @@
pandoc "out.md" -o output1310.pdf \
--pdf-engine=xelatex \
-V geometry:margin=2.5cm \
-V fontsize=12pt \
-V mainfont="Times New Roman" \
-V colorlinks=true \
-V linkcolor=blue\

View File

@@ -1,11 +1,7 @@
from faster_whisper import WhisperModel from faster_whisper import WhisperModel
class FasterWhisper: class FasterWhisper:
def recognize(self, model, device, compute_type, def recognize(self, model, device, compute_type, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText):
audioFile, beamSize, vadFilter,
minSilenceDurationMs, speechPadMs,
temp0, temp1, temp2, wordTimestamps,
noSpeechThreshold, conditionOnPreviousText):
model = WhisperModel(model, device=device, compute_type=compute_type) # Задаем модель model = WhisperModel(model, device=device, compute_type=compute_type) # Задаем модель
segments, _ = model.transcribe( # Распознаем текст segments, _ = model.transcribe( # Распознаем текст
@@ -26,7 +22,6 @@ class FasterWhisper:
for seg in segments: for seg in segments:
text += f"[{self.format_timestamp(seg.start)} -> {self.format_timestamp(seg.end)}] {seg.text}" + '\n' text += f"[{self.format_timestamp(seg.start)} -> {self.format_timestamp(seg.end)}] {seg.text}" + '\n'
print(f"[{self.format_timestamp(seg.start)} -> {self.format_timestamp(seg.end)}] {seg.text}")
return text return text

View File

@@ -3,9 +3,8 @@ from services.llm_providers.ionet_provider import IoNetProvider
from services.llm_providers.gemini_provider import GeminiProvider from services.llm_providers.gemini_provider import GeminiProvider
from services.llm_providers.gpt4free_provider import Gpt4FreeProvider from services.llm_providers.gpt4free_provider import Gpt4FreeProvider
from services.llm_providers.base_provider import BaseLLMProvider from services.llm_providers.base_provider import BaseLLMProvider
from services.llm_providers.custom_provider import CustomProvider
def get_llm_provider(provider_name: str, api_key: str | None = None, base_url: str | None = None) -> BaseLLMProvider: def get_llm_provider(provider_name: str, api_key: str | None) -> BaseLLMProvider:
""" """
Фабричная функция для получения экземпляра провайдера LLM. Фабричная функция для получения экземпляра провайдера LLM.
""" """
@@ -19,9 +18,5 @@ def get_llm_provider(provider_name: str, api_key: str | None = None, base_url: s
return GeminiProvider(api_key) return GeminiProvider(api_key)
elif provider_name == 'gpt4free': elif provider_name == 'gpt4free':
return Gpt4FreeProvider() return Gpt4FreeProvider()
elif provider_name == 'Custom':
if not base_url:
raise ValueError("Base URL обязателен для Custom провайдера")
return CustomProvider(api_key, base_url) # base_url будет установлен позже
else: else:
raise ValueError(f"Неизвестный провайдер: {provider_name}") raise ValueError(f"Неизвестный провайдер: {provider_name}")

View File

@@ -1,44 +0,0 @@
import requests
from .base_provider import BaseLLMProvider
class CustomProvider(BaseLLMProvider):
def __init__(self, api_key: str | None = None, base_url: str | None = None):
super().__init__(api_key)
self.base_url = base_url or 'http://127.0.0.1:1234/v1/'
# ГАРАНТИРУЕМ наличие слеша в конце URL
if not self.base_url.endswith('/'):
self.base_url += '/'
def generate(self, model: str, system_prompt: str, user_prompt: str, temp: float):
url = f"{self.base_url}chat/completions"
# Некоторые Custom провайдеры (как vLLM или Ollama) могут требовать API Key, даже если он фиктивный
headers = {"Content-Type": "application/json"}
if self.api_key:
headers["Authorization"] = f"Bearer {self.api_key}"
data = {
"model": model,
"messages": [
{'role': 'system', 'content': system_prompt},
{'role': 'user', 'content': user_prompt},
],
"temperature": temp,
"stream": False
}
try:
response = requests.post(url, headers=headers, json=data, timeout=300)
response.raise_for_status()
result = response.json()
# Обработка разных форматов ответа (на всякий случай)
if 'choices' in result and len(result['choices']) > 0:
text = str(result['choices'][0]['message']['content'])
return text, text
else:
return f"Неожиданный ответ от сервера: {result}", str(result)
except requests.exceptions.RequestException as e:
return f"Ошибка при запросе к CustomProvider API: {e}", f"Ошибка: {e}"

View File

@@ -41,7 +41,7 @@ class GeminiProvider(BaseLLMProvider):
try: try:
# 4. Отправляем POST-запрос с данными и настройками прокси # 4. Отправляем POST-запрос с данными и настройками прокси
response = requests.post(api_url, json=data, proxies=proxies, timeout=400) response = requests.post(api_url, json=data, proxies=proxies, timeout=90)
# Проверяем, не вернул ли сервер ошибку (например, 4xx или 5xx) # Проверяем, не вернул ли сервер ошибку (например, 4xx или 5xx)
response.raise_for_status() response.raise_for_status()

View File

@@ -3,13 +3,6 @@ from g4f.client import Client
from .base_provider import BaseLLMProvider from .base_provider import BaseLLMProvider
class Gpt4FreeProvider(BaseLLMProvider): class Gpt4FreeProvider(BaseLLMProvider):
"""
Провайдер для работы с моделью GPT через библиотеку gpt4free.
Этот класс реализует интерфейс BaseLLMProvider и предоставляет возможность
взаимодействия с различными LLM через сервис gpt4free, который не требует
API ключа для работы.
"""
# gpt4free не требует API ключа # gpt4free не требует API ключа
def __init__(self, api_key: str | None = None): def __init__(self, api_key: str | None = None):
super().__init__(api_key) super().__init__(api_key)
@@ -17,18 +10,6 @@ class Gpt4FreeProvider(BaseLLMProvider):
def generate(self, model: str, system_prompt: str, user_prompt: str, temp: float): def generate(self, model: str, system_prompt: str, user_prompt: str, temp: float):
"""
Генерирует ответ от модели GPT с использованием gpt4free.
Args:
model (str): Название модели для генерации ответа
system_prompt (str): Системное сообщение для контекста
user_prompt (str): Пользовательский запрос
temp (float): Температура генерации ( controls randomness of responses)
Returns:
tuple: Кортеж из двух одинаковых строк - сгенерированного ответа и его копии
"""
# temp в g4f может работать не для всех внутренних провайдеров # temp в g4f может работать не для всех внутренних провайдеров
try: try:
response = self.client.chat.completions.create( response = self.client.chat.completions.create(

View File

@@ -1,34 +1,23 @@
import requests import openai
from .base_provider import BaseLLMProvider from .base_provider import BaseLLMProvider
class IoNetProvider(BaseLLMProvider): class IoNetProvider(BaseLLMProvider):
def __init__(self, api_key: str): def __init__(self, api_key: str):
super().__init__(api_key) super().__init__(api_key)
self.base_url = 'https://api.intelligence.io.solutions/api/v1' self.client = openai.OpenAI(
api_key=self.api_key,
base_url='https://api.intelligence.io.solutions/api/v1/'
)
def generate(self, model: str, system_prompt: str, user_prompt: str, temp: float): def generate(self, model: str, system_prompt: str, user_prompt: str, temp: float):
url = f"{self.base_url}/chat/completions" response = self.client.chat.completions.create(
headers = { model=model,
"Content-Type": "application/json", messages=[
"Authorization": f"Bearer {self.api_key}"
}
data = {
"model": model,
"messages": [
{'role': 'system', 'content': system_prompt}, {'role': 'system', 'content': system_prompt},
{'role': 'user', 'content': user_prompt}, {'role': 'user', 'content': user_prompt},
], ],
"temperature": temp temperature=temp,
} stream=False
)
try: text = str(response.choices[0].message.content)
response = requests.post(url, headers=headers, json=data)
response.raise_for_status()
result = response.json()
text = str(result['choices'][0]['message']['content'])
return text, text # Возвращаем как чистый текст, так и Markdown return text, text # Возвращаем как чистый текст, так и Markdown
except requests.exceptions.RequestException as e:
raise Exception(f"Ошибка при запросе к IO.net API: {e}")