Compare commits
1 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d231707572 |
91
app.py
91
app.py
@@ -1,9 +1,17 @@
|
||||
import gradio as gr
|
||||
|
||||
# Загрузка параметров конфигурации
|
||||
from config import *
|
||||
|
||||
# Подгрузка сервисов
|
||||
from services.llm_factory import get_llm_provider
|
||||
from services.fasterWhisper import FasterWhisper
|
||||
|
||||
|
||||
|
||||
# Загрузка доп. модулей
|
||||
from handlers.gradioHandler import GradioHandlers
|
||||
from handlers.fileHandlers import FileHandlers
|
||||
from services.fasterWhisper import FasterWhisper
|
||||
from handlers.convertMdToPdf import ConvertMdToPdf
|
||||
from handlers.glueAudio import GlueAudio
|
||||
|
||||
@@ -11,9 +19,16 @@ gh = GradioHandlers(get_llm_provider, ConvertMdToPdf, FileHandlers, FasterWhispe
|
||||
|
||||
def main():
|
||||
with gr.Blocks() as demo:
|
||||
gr.HTML('<div align=center><h1>Faster Whisper WebUI</h1></div>')
|
||||
gr.HTML('''
|
||||
<div align=center>
|
||||
<h1>
|
||||
Faster Whisper WebUI
|
||||
</h1>
|
||||
</div>
|
||||
''')
|
||||
|
||||
with gr.Row():
|
||||
# Вкладка с основным взаимодействием
|
||||
with gr.Tab('Actions'):
|
||||
isPipelineEnabledCheckbox = gr.Checkbox(label='is pipeline enabled', value=True, interactive=True)
|
||||
|
||||
@@ -23,102 +38,104 @@ def main():
|
||||
audioFiles = gr.Files(label='Load audio for transcribe', type="filepath")
|
||||
images = gr.Files(label='Upload images', file_types=['image'])
|
||||
recognizeBtn = gr.Button('recognize and integrate', variant='primary')
|
||||
|
||||
with gr.Accordion(label='Recognized text'):
|
||||
recognizedText = gr.TextArea(label='')
|
||||
|
||||
with gr.Accordion(label='LLM'):
|
||||
with gr.Column():
|
||||
refineTextBtn = gr.Button('refine text', variant='secondary', interactive=False)
|
||||
|
||||
with gr.Accordion(label='Refined text raw'):
|
||||
refinedText = gr.Textbox(label='', show_copy_button=True)
|
||||
|
||||
with gr.Accordion(label='Refined text md formated'):
|
||||
refinedTextMD = gr.Markdown(label='')
|
||||
|
||||
# Вкладка с настройками
|
||||
with gr.Tab('Settings'):
|
||||
with gr.Column():
|
||||
# Первое поле на всю ширину в акордионе настроек
|
||||
with gr.Accordion('File settings'):
|
||||
saveFileCheckbox = gr.Checkbox(label='save file', value=True, interactive=True)
|
||||
filename = gr.Textbox(label='Output filename', value='output.md', interactive=True)
|
||||
filename = gr.Textbox(label='Output filename', value='output.txt', interactive=True)
|
||||
filenamePdf = gr.Textbox(label='Output filename for pdf', value='output.pdf', interactive=True)
|
||||
|
||||
# Акордион настроек faster whisper
|
||||
with gr.Accordion(label='Faster whisper settings'):
|
||||
with gr.Row():
|
||||
# Левая колонка в акордионе
|
||||
with gr.Column():
|
||||
device = gr.Dropdown(label='Device', choices=DEVICES, value=DEVICES[1], interactive=True)
|
||||
compute_type = gr.Dropdown(label='compute_type', choices=COMPUTE_TYPE, value=COMPUTE_TYPE[0], interactive=True)
|
||||
fastWhisperModel = gr.Dropdown(label='Model', choices=FAST_WHISPER_MODELS, value=FAST_WHISPER_MODELS[11], interactive=True)
|
||||
|
||||
beamSize = gr.Number(label='beam_size', value=8, interactive=True)
|
||||
noSpeechThreshold = gr.Number(label='no_speech_threshold', value=0.5, interactive=True)
|
||||
vadFilter = gr.Checkbox(label='vad_filter', value=True, interactive=True)
|
||||
wordTimestamps = gr.Checkbox(label='word_timestamps', value=True, interactive=True)
|
||||
conditionOnPreviousText = gr.Checkbox(label='condition_on_previous_text', value=False, interactive=True)
|
||||
|
||||
# Правая колонка в акордионе
|
||||
with gr.Column():
|
||||
with gr.Accordion(label='Vad parameters'):
|
||||
minSilenceDurationMs = gr.Number(label='min_silence_duration_ms', value=300, interactive=True)
|
||||
speechPadMs = gr.Number(label='speech_pad_ms', value=200, interactive=True)
|
||||
|
||||
with gr.Accordion(label='Temperature'):
|
||||
temp0 = gr.Number(label='temp_0', value=0.0, interactive=True)
|
||||
temp1 = gr.Number(label='temp_1', value=0.2, interactive=True)
|
||||
temp2 = gr.Number(label='temp_2', value=0.4, interactive=True)
|
||||
|
||||
# Нижний акордион настроек для api ключа llm
|
||||
with gr.Accordion(label='LLM settings'):
|
||||
apiKey = gr.Textbox(label='API key (required for io.net, Gemini)', value=DEFAULT_API_KEY, interactive=True)
|
||||
|
||||
with gr.Accordion(label='System prompt'):
|
||||
systemPrompt = gr.Textbox(label='', value=DEFAULT_SYSTEM_PROMPT, interactive=True)
|
||||
|
||||
with gr.Row():
|
||||
llmProvider = gr.Dropdown(label='LLM Provider', choices=LLM_PROVIDERS, value=LLM_PROVIDERS[0], interactive=True)
|
||||
llmModel = gr.Dropdown(label='Models', choices=LLM_MODELS[LLM_PROVIDERS[0]], value=LLM_MODELS[LLM_PROVIDERS[0]][1], interactive=True)
|
||||
# ВЫБОР ПРОВАЙДЕРА
|
||||
llmProvider = gr.Dropdown(
|
||||
label='LLM Provider',
|
||||
choices=LLM_PROVIDERS,
|
||||
value=LLM_PROVIDERS[0],
|
||||
interactive=True
|
||||
)
|
||||
# СПИСОК МОДЕЛЕЙ (теперь зависит от провайдера)
|
||||
llmModel = gr.Dropdown(
|
||||
label='Models',
|
||||
choices=LLM_MODELS[LLM_PROVIDERS[0]], # Модели для провайдера по умолчанию
|
||||
value=LLM_MODELS[LLM_PROVIDERS[0]][1],
|
||||
interactive=True
|
||||
)
|
||||
llmTemperature = gr.Number(label='Temperature', value=0.8, interactive=True)
|
||||
|
||||
# Настройки Custom провайдера
|
||||
with gr.Accordion(label='Custom Provider Settings', open=True):
|
||||
customBaseUrl = gr.Textbox(
|
||||
label='Base URL',
|
||||
value='http://127.0.0.1:1234/v1/',
|
||||
interactive=True,
|
||||
visible=False # Скрыто по умолчанию
|
||||
)
|
||||
|
||||
# Обработчики событий
|
||||
isPipelineEnabledCheckbox.change(gh.updateButton, inputs=[isPipelineEnabledCheckbox], outputs=refineTextBtn)
|
||||
saveFileCheckbox.change(gh.updateTextbox, inputs=saveFileCheckbox, outputs=filename)
|
||||
saveFileCheckbox.change(gh.updateTextbox, inputs=saveFileCheckbox, outputs=filenamePdf)
|
||||
|
||||
recognizeBtn.click(
|
||||
gh.handleRecognizeBtn,
|
||||
inputs=[audioFiles, fastWhisperModel, device, compute_type, beamSize,
|
||||
vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2,
|
||||
wordTimestamps, noSpeechThreshold, conditionOnPreviousText, gr.State(GLUED_AUDIO_FILENAME), gr.State(OUTPUT_PATH)],
|
||||
outputs=[recognizedText],
|
||||
)
|
||||
recognizeBtn.click(gh.handleRecognizeBtn, outputs=[recognizedText], inputs=[audioFiles, fastWhisperModel, device, compute_type, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText, gr.State(GLUED_AUDIO_FILENAME), gr.State(OUTPUT_PATH)])
|
||||
|
||||
# --- ИСПРАВЛЕНИЕ: ДОБАВЛЕН customBaseUrl В INPUTS ---
|
||||
# Если пайплайн включен то тогда делаем автоматически
|
||||
# автоматический пайплайн
|
||||
recognizedText.change(
|
||||
gh.generateByCondition,
|
||||
inputs=[apiKey, llmProvider, llmModel, systemPrompt, recognizedText, llmTemperature,
|
||||
isPipelineEnabledCheckbox, gr.State("change"), saveFileCheckbox, filename, filenamePdf, gr.State(OUTPUT_PATH), customBaseUrl],
|
||||
inputs=[apiKey, llmProvider, llmModel, systemPrompt, recognizedText, llmTemperature, isPipelineEnabledCheckbox, gr.State("change"), saveFileCheckbox, filename, filenamePdf, gr.State(OUTPUT_PATH)],
|
||||
outputs=[refinedText, refinedTextMD]
|
||||
)
|
||||
|
||||
# Обновление выпадающего списка моделей и поля API ключа
|
||||
|
||||
|
||||
# ручной запуск по кнопке
|
||||
llmProvider.change(
|
||||
gh.update_model_dropdown,
|
||||
inputs=llmProvider,
|
||||
outputs=[llmModel, apiKey]
|
||||
outputs=llmModel
|
||||
)
|
||||
|
||||
# Переключение видимости URL для Custom провайдера
|
||||
llmProvider.change(
|
||||
fn=gh.toggle_custom_url,
|
||||
inputs=llmProvider,
|
||||
outputs=[customBaseUrl]
|
||||
)
|
||||
|
||||
refineTextBtn.click(
|
||||
gh.generateByCondition,
|
||||
inputs=[apiKey, llmProvider, llmModel, systemPrompt, recognizedText, llmTemperature,
|
||||
isPipelineEnabledCheckbox, gr.State("click"), saveFileCheckbox, filename, filenamePdf, gr.State(OUTPUT_PATH), customBaseUrl],
|
||||
inputs=[apiKey, llmProvider, llmModel, systemPrompt, recognizedText, llmTemperature, isPipelineEnabledCheckbox, gr.State("click"), saveFileCheckbox, filename, filenamePdf, gr.State(OUTPUT_PATH)],
|
||||
outputs=[refinedText, refinedTextMD]
|
||||
)
|
||||
|
||||
|
||||
78
config.py
78
config.py
@@ -8,13 +8,10 @@ DEVICES = ['cpu', 'cuda']
|
||||
COMPUTE_TYPE = ['auto', 'int8', 'float16', 'float32']
|
||||
|
||||
# Стандартный API ключ
|
||||
IO_API_KEY=os.getenv('IO_API_KEY')
|
||||
GEMINI_API_KEY=os.getenv('GEMINI_API_KEY')
|
||||
DEFAULT_API_KEY=IO_API_KEY
|
||||
|
||||
DEFAULT_API_KEY=os.getenv('API_KEY')
|
||||
|
||||
# Словарь провайдеров и их моделей
|
||||
LLM_PROVIDERS = ['io.net', 'Gemini', 'gpt4free', 'Custom']
|
||||
LLM_PROVIDERS = ['io.net', 'Gemini', 'gpt4free']
|
||||
LLM_MODELS = {
|
||||
'io.net': [
|
||||
'openai/gpt-oss-120b', 'Qwen/Qwen3-235B-A22B-Thinking-2507',
|
||||
@@ -30,7 +27,7 @@ LLM_MODELS = {
|
||||
'gemini-2.5-flash',
|
||||
'gemini-2.5-flash-lite'
|
||||
],
|
||||
'gpt4free': [
|
||||
'gpt4free': [ # Модели могут меняться, проверьте документацию g4f
|
||||
'default',
|
||||
'gpt-4',
|
||||
'sonar-reasoning',
|
||||
@@ -41,14 +38,7 @@ LLM_MODELS = {
|
||||
'gpt-4o-mini',
|
||||
'deepseek-r1',
|
||||
'PollinationsAI:gpt-5-nano'
|
||||
],
|
||||
'Custom': [
|
||||
'qwen/qwen3-vl-30b',
|
||||
'qwen/qwen3-coder-30b',
|
||||
'openai/gpt-oss-20b',
|
||||
'qwen3-vl-8b-thinking',
|
||||
'qwen/qwen3-vl-8b',
|
||||
],
|
||||
]
|
||||
}
|
||||
|
||||
# Задаем выходную директорию
|
||||
@@ -56,38 +46,42 @@ OUTPUT_PATH='outputs'
|
||||
|
||||
GLUED_AUDIO_FILENAME='glued.mp3'
|
||||
|
||||
DEFAULT_SYSTEM_PROMPT='''You are a diligent university student who has recorded a lecture as an audio file and later transcribed it into raw text.
|
||||
Your task is to rewrite this unstructured transcript into a clear, logically organized, and detailed lecture summary (lecture notes).
|
||||
DEFAULT_SYSTEM_PROMPT='''
|
||||
You are a smart university student creating easy-to-understand study notes summary of lesson for a classmate who is a beginner. Your source is a raw text/audio transcript.
|
||||
|
||||
Guidelines:
|
||||
1. Structure:
|
||||
- Organize the text into a hierarchy of sections and subsections.
|
||||
- Use headings, bullet points, or numbering where appropriate.
|
||||
- Present the material in a logical flow (from introduction → main points → details → examples → conclusion).
|
||||
GOAL: rewrite the information into a clear, structured summary in RUSSIAN.
|
||||
|
||||
2. Clarity & Cohesion:
|
||||
- Remove filler words, repetitions, and irrelevant fragments.
|
||||
- Rewrite incomplete sentences into full, grammatically correct sentences.
|
||||
- Ensure smooth transitions between topics, making the summary feel continuous and well-connected.
|
||||
KEY RULES FOR CONTENT:
|
||||
1. **Logical Structure:** Use Markdown headers (#, ##), bullet points, and short paragraphs.
|
||||
2. **No "Water":** Remove filler words. Keep only practical information.
|
||||
3. **Student Tone:** Write naturally, as if sharing notes with a friend. Avoid robotic phrases like "It is important to note".
|
||||
|
||||
3. Depth & Detail:
|
||||
- Capture all important concepts, definitions, examples, and explanations from the lecture.
|
||||
- Expand shorthand or fragmented thoughts into full, precise explanations.
|
||||
- Where appropriate, rephrase or clarify confusing passages for better understanding.
|
||||
KEY RULES FOR LATEX (CRITICAL FOR PYLATEXENC):
|
||||
1. **Math Mode:** ANY variable (like t, L, C), number in a formula, or equation MUST be wrapped in dollar signs `$`.
|
||||
* BAD: i(t) = i_pr + i_sv
|
||||
* GOOD: $i(t) = i_{pr} + i_{sv}$
|
||||
2. **Subscripts:** Always use curly braces `{}` for subscripts longer than one character.
|
||||
* BAD: $i_pr$
|
||||
* GOOD: $i_{pr}$ (or $i_{пр}$ if using cyrillic)
|
||||
3. **Symbols:** Use standard LaTeX commands for symbols.
|
||||
* Arrow: use `\to` (e.g., $t \to \infty$).
|
||||
* Infinity: use `\infty`.
|
||||
* Multiplication: use `\cdot` or just space.
|
||||
4. **Consistency:** Never leave a mathematical symbol as plain text. If you mention "current i", write "ток $i$".
|
||||
|
||||
4. Accuracy:
|
||||
- Preserve the lecturer’s original meaning, intent, and terminology.
|
||||
- Avoid adding personal opinions or new information that was not in the lecture.
|
||||
EXAMPLE OF DESIRED OUTPUT FORMAT:
|
||||
# Тема лекции
|
||||
## Основные понятия
|
||||
* **Переходный процесс** — это когда цепь перестраивается с одного режима на другой (например, щелкнули выключателем).
|
||||
* Математически это описывается дифференциальными уравнениями. Порядок уравнения = количеству реактивных элементов ($L$ и $C$).
|
||||
|
||||
5. Style:
|
||||
- Write in a formal, academic tone suitable for study notes.
|
||||
- Aim for readability: concise sentences, but thorough coverage of concepts.
|
||||
- Use emphasis (e.g., bold or italic text) only when it improves comprehension.
|
||||
## Классический метод
|
||||
Решение ищется в виде суммы двух частей:
|
||||
$$i(t) = i_{pr} + i_{sv}$$
|
||||
|
||||
Final Output: A cohesive, detailed, and well-structured lecture summary, suitable for later studying and revision.
|
||||
Use only russian language!
|
||||
USE LATEX IN DOLLAR SIGN ($)!
|
||||
EXTRA BIG LENTH OF CONSPECT!
|
||||
MAKE AS LONG AS POSIBLE AND AS BE GOOD!
|
||||
1. **Принужденная составляющая** ($i_{pr}$) — это режим, который установится в будущем, когда все успокоится ($t \to \infty$).
|
||||
2. **Свободная составляющая** ($i_{sv}$) — это то, что происходит "само по себе" из-за энергии, запасенной в $L$ и $C$.
|
||||
|
||||
***
|
||||
STRICTLY FOLLOW THESE FORMATTING RULES. OUTPUT IN RUSSIAN.
|
||||
'''
|
||||
|
||||
|
||||
@@ -1,61 +1,11 @@
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
|
||||
from pydub import AudioSegment
|
||||
|
||||
class GlueAudio():
|
||||
def glue(self, audio_files: list, output_path: str, output_filename: str) -> Path:
|
||||
"""
|
||||
Склеивает аудиофайлы РАЗНЫХ форматов с помощью FFmpeg и filter_complex.
|
||||
Это универсальный и эффективный по памяти метод.
|
||||
def glue(self, audioFiles):
|
||||
glued = AudioSegment.empty()
|
||||
|
||||
Args:
|
||||
audio_files (list): Список путей к исходным аудиофайлам.
|
||||
output_path (str): Директория для сохранения итогового файла.
|
||||
output_filename (str): Имя итогового склеенного файла.
|
||||
for audioFile in audioFiles:
|
||||
audio = AudioSegment.from_file(audioFile)
|
||||
glued += audio
|
||||
|
||||
Returns:
|
||||
Path: Путь к созданному склеенному файлу.
|
||||
"""
|
||||
output_dir = Path(output_path)
|
||||
output_dir.mkdir(parents=True, exist_ok=True)
|
||||
final_audio_path = output_dir / output_filename
|
||||
|
||||
if not audio_files:
|
||||
raise ValueError("Список аудиофайлов для склейки пуст.")
|
||||
|
||||
# 1. Формируем часть команды с входными файлами (-i file1 -i file2 ...)
|
||||
input_args = []
|
||||
for file_path in audio_files:
|
||||
input_args.extend(['-i', str(Path(file_path).resolve())])
|
||||
|
||||
# 2. Формируем строку для filter_complex
|
||||
num_files = len(audio_files)
|
||||
stream_specifiers = "".join([f"[{i}:a]" for i in range(num_files)])
|
||||
filter_complex_str = f"{stream_specifiers}concat=n={num_files}:v=0:a=1[outa]"
|
||||
|
||||
# 3. Собираем полную команду
|
||||
command = [
|
||||
'ffmpeg',
|
||||
*input_args, # Распаковываем список входных файлов
|
||||
'-filter_complex', filter_complex_str,
|
||||
'-map', '[outa]',
|
||||
'-c:a', 'libmp3lame',
|
||||
'-q:a', '2',
|
||||
str(final_audio_path),
|
||||
'-y'
|
||||
]
|
||||
|
||||
try:
|
||||
# 4. Выполняем команду
|
||||
print(f"Выполнение команды FFmpeg: {' '.join(command)}")
|
||||
subprocess.run(command, check=True, capture_output=True, text=True)
|
||||
print("FFmpeg успешно завершил склейку.")
|
||||
|
||||
except FileNotFoundError:
|
||||
raise FileNotFoundError("FFmpeg не найден. Убедитесь, что он установлен и доступен в системной переменной PATH.")
|
||||
except subprocess.CalledProcessError as e:
|
||||
print("Ошибка при выполнении FFmpeg!")
|
||||
print("Stderr:", e.stderr)
|
||||
raise RuntimeError(f"Ошибка FFmpeg при склейке файлов: {e.stderr}")
|
||||
|
||||
return final_audio_path
|
||||
return glued
|
||||
|
||||
@@ -1,52 +1,34 @@
|
||||
from config import LLM_MODELS, GEMINI_API_KEY, IO_API_KEY
|
||||
from config import LLM_MODELS # Импортируем словарь моделей
|
||||
import gradio as gr
|
||||
|
||||
|
||||
class GradioHandlers:
|
||||
def __init__(self, llm_factory, ConvertMdToPdf, FileHandlers, FasterWhisper, GlueAudio):
|
||||
# Объект для работы с файлами
|
||||
self.fh = FileHandlers()
|
||||
self.ga = GlueAudio()
|
||||
self.ConvertMdToPdf = ConvertMdToPdf()
|
||||
self.FasterWhisper = FasterWhisper()
|
||||
self.llm_factory = llm_factory
|
||||
self.llm_factory = llm_factory # Сохраняем фабрику
|
||||
|
||||
def handleRecognizeBtn(self, audioFiles, model, device, compute_type, beamSize, vadFilter,
|
||||
minSilenceDurationMs, speechPadMs, temp0, temp1, temp2,
|
||||
wordTimestamps, noSpeechThreshold, conditionOnPreviousText, filename, outPath):
|
||||
def handleRecognizeBtn(self, audioFiles, model, device, compute_type, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText, filename, outPath):
|
||||
audioFile = self.ga.glue(audioFiles)
|
||||
file = self.fh.saveFile(filename, audioFile, outPath)
|
||||
|
||||
return self.FasterWhisper.recognize(model, device, compute_type, file, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText)
|
||||
|
||||
# Функция улучшения текста
|
||||
def generateByCondition(self, api_key, llm_provider, llm_model, system_prompt, recognized_text, llm_temperature, is_pipeline_enabled, trigger, isSaveFile, filename, filenamePdf, output_path):
|
||||
try:
|
||||
glued_audio_path = self.ga.glue(
|
||||
audio_files=[f.name for f in audioFiles],
|
||||
output_path=outPath,
|
||||
output_filename=filename
|
||||
)
|
||||
except (FileNotFoundError, RuntimeError) as e:
|
||||
gr.Warning(str(e))
|
||||
return ""
|
||||
|
||||
return self.FasterWhisper.recognize(model, device, compute_type, str(glued_audio_path), beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText)
|
||||
|
||||
# Добавил аргумент custom_base_url в конец
|
||||
def generateByCondition(self, api_key, llm_provider,
|
||||
llm_model, system_prompt, recognized_text,
|
||||
llm_temperature, is_pipeline_enabled, trigger,
|
||||
isSaveFile, filename, filenamePdf, output_path, custom_base_url):
|
||||
|
||||
try:
|
||||
if llm_provider == "Custom":
|
||||
# Передаем base_url только для Custom
|
||||
provider = self.llm_factory(llm_provider, api_key, base_url=custom_base_url)
|
||||
else:
|
||||
# Получаем нужный провайдер через фабрику
|
||||
provider = self.llm_factory(llm_provider, api_key)
|
||||
except ValueError as e:
|
||||
gr.Warning(str(e))
|
||||
return gr.skip(), gr.skip()
|
||||
# Если API ключ не предоставлен для нужного провайдера, выводим ошибку
|
||||
self.gr.Warning(str(e))
|
||||
return self.gr.skip(), self.gr.skip()
|
||||
|
||||
def process():
|
||||
# Добавлена обработка ошибок генерации
|
||||
try:
|
||||
result, md = provider.generate(llm_model, system_prompt, recognized_text, llm_temperature)
|
||||
except Exception as e:
|
||||
raise gr.Error(f"Ошибка генерации LLM: {e}")
|
||||
|
||||
pdf, unicodeText = self.ConvertMdToPdf.convertLatexToText(md)
|
||||
if isSaveFile:
|
||||
self.fh.saveFile(filenamePdf, pdf, output_path)
|
||||
@@ -58,30 +40,28 @@ class GradioHandlers:
|
||||
|
||||
return gr.skip(), gr.skip()
|
||||
|
||||
# НОВАЯ ФУНКЦИЯ для обновления списка моделей
|
||||
def update_model_dropdown(self, provider):
|
||||
"""
|
||||
Вызывается при изменении llmProvider.
|
||||
Возвращает обновленный компонент Dropdown для моделей.
|
||||
"""
|
||||
# Получаем список моделей для выбранного провайдера
|
||||
models = LLM_MODELS.get(provider, [])
|
||||
|
||||
# Выбираем первое значение по умолчанию, если список не пуст
|
||||
default_value = models[0] if models else None
|
||||
|
||||
# Обновляем список моделей и настройки поля API Key
|
||||
if provider == 'io.net':
|
||||
return gr.update(choices=models, value=default_value), gr.update(label='API key', value=IO_API_KEY, interactive=True, visible=True)
|
||||
if provider == 'Gemini':
|
||||
return gr.update(choices=models, value=default_value), gr.update(label='API key', value=GEMINI_API_KEY, interactive=True, visible=True)
|
||||
if provider == 'gpt4free':
|
||||
return gr.update(choices=models, value=default_value), gr.update(label='API key (not required)', value="", interactive=False, visible=True)
|
||||
if provider == "Custom":
|
||||
return gr.update(choices=models, value=default_value), gr.update(label='API key (optional)', value="", interactive=True, visible=True) # Для Custom ключ может понадобиться
|
||||
# Возвращаем обновленный компонент. Используем 'gr' напрямую.
|
||||
return gr.update(choices=models, value=default_value)
|
||||
|
||||
# Функция для динамического обновления кнопки
|
||||
def updateButton(self, isChecked):
|
||||
variant = 'secondary' if isChecked else 'primary'
|
||||
if not isChecked:
|
||||
variant = 'primary'
|
||||
else:
|
||||
variant = 'secondary'
|
||||
return gr.update(interactive=not isChecked, variant=variant)
|
||||
|
||||
def toggle_custom_url(self, provider):
|
||||
"""Показывает поле Base URL только если выбран Custom"""
|
||||
return gr.update(visible=(provider == 'Custom'))
|
||||
|
||||
def update_custom_url(self, base_url):
|
||||
return None
|
||||
|
||||
def updateTextbox(self, isChecked):
|
||||
return gr.update(visible=isChecked)
|
||||
@@ -1,21 +0,0 @@
|
||||
import os
|
||||
from PIL import Image, ExifTags
|
||||
import subprocess
|
||||
import json
|
||||
from datetime import datetime
|
||||
|
||||
class MetadataHandler:
|
||||
def get_image_timestamp(self, image_path: str) -> datetime | None:
|
||||
"""Извлекает метку времени из метаданных изображения, если она доступна."""
|
||||
try:
|
||||
image = Image.open(image_path)
|
||||
exif_data = image._getexif()
|
||||
if exif_data:
|
||||
for tag, value in exif_data.items():
|
||||
decoded_tag = ExifTags.TAGS.get(tag, tag)
|
||||
if decoded_tag == 'DateTimeOriginal':
|
||||
return value
|
||||
return None
|
||||
except Exception as e:
|
||||
print(f"Error extracting metadata from image: {e}")
|
||||
return None
|
||||
@@ -1,7 +0,0 @@
|
||||
pandoc "out.md" -o output1310.pdf \
|
||||
--pdf-engine=xelatex \
|
||||
-V geometry:margin=2.5cm \
|
||||
-V fontsize=12pt \
|
||||
-V mainfont="Times New Roman" \
|
||||
-V colorlinks=true \
|
||||
-V linkcolor=blue\
|
||||
@@ -1,11 +1,7 @@
|
||||
from faster_whisper import WhisperModel
|
||||
|
||||
class FasterWhisper:
|
||||
def recognize(self, model, device, compute_type,
|
||||
audioFile, beamSize, vadFilter,
|
||||
minSilenceDurationMs, speechPadMs,
|
||||
temp0, temp1, temp2, wordTimestamps,
|
||||
noSpeechThreshold, conditionOnPreviousText):
|
||||
def recognize(self, model, device, compute_type, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText):
|
||||
model = WhisperModel(model, device=device, compute_type=compute_type) # Задаем модель
|
||||
|
||||
segments, _ = model.transcribe( # Распознаем текст
|
||||
@@ -26,7 +22,6 @@ class FasterWhisper:
|
||||
|
||||
for seg in segments:
|
||||
text += f"[{self.format_timestamp(seg.start)} -> {self.format_timestamp(seg.end)}] {seg.text}" + '\n'
|
||||
print(f"[{self.format_timestamp(seg.start)} -> {self.format_timestamp(seg.end)}] {seg.text}")
|
||||
|
||||
return text
|
||||
|
||||
|
||||
@@ -3,9 +3,8 @@ from services.llm_providers.ionet_provider import IoNetProvider
|
||||
from services.llm_providers.gemini_provider import GeminiProvider
|
||||
from services.llm_providers.gpt4free_provider import Gpt4FreeProvider
|
||||
from services.llm_providers.base_provider import BaseLLMProvider
|
||||
from services.llm_providers.custom_provider import CustomProvider
|
||||
|
||||
def get_llm_provider(provider_name: str, api_key: str | None = None, base_url: str | None = None) -> BaseLLMProvider:
|
||||
def get_llm_provider(provider_name: str, api_key: str | None) -> BaseLLMProvider:
|
||||
"""
|
||||
Фабричная функция для получения экземпляра провайдера LLM.
|
||||
"""
|
||||
@@ -19,9 +18,5 @@ def get_llm_provider(provider_name: str, api_key: str | None = None, base_url: s
|
||||
return GeminiProvider(api_key)
|
||||
elif provider_name == 'gpt4free':
|
||||
return Gpt4FreeProvider()
|
||||
elif provider_name == 'Custom':
|
||||
if not base_url:
|
||||
raise ValueError("Base URL обязателен для Custom провайдера")
|
||||
return CustomProvider(api_key, base_url) # base_url будет установлен позже
|
||||
else:
|
||||
raise ValueError(f"Неизвестный провайдер: {provider_name}")
|
||||
@@ -1,44 +0,0 @@
|
||||
import requests
|
||||
from .base_provider import BaseLLMProvider
|
||||
|
||||
class CustomProvider(BaseLLMProvider):
|
||||
def __init__(self, api_key: str | None = None, base_url: str | None = None):
|
||||
super().__init__(api_key)
|
||||
self.base_url = base_url or 'http://127.0.0.1:1234/v1/'
|
||||
|
||||
# ГАРАНТИРУЕМ наличие слеша в конце URL
|
||||
if not self.base_url.endswith('/'):
|
||||
self.base_url += '/'
|
||||
|
||||
def generate(self, model: str, system_prompt: str, user_prompt: str, temp: float):
|
||||
url = f"{self.base_url}chat/completions"
|
||||
|
||||
# Некоторые Custom провайдеры (как vLLM или Ollama) могут требовать API Key, даже если он фиктивный
|
||||
headers = {"Content-Type": "application/json"}
|
||||
if self.api_key:
|
||||
headers["Authorization"] = f"Bearer {self.api_key}"
|
||||
|
||||
data = {
|
||||
"model": model,
|
||||
"messages": [
|
||||
{'role': 'system', 'content': system_prompt},
|
||||
{'role': 'user', 'content': user_prompt},
|
||||
],
|
||||
"temperature": temp,
|
||||
"stream": False
|
||||
}
|
||||
|
||||
try:
|
||||
response = requests.post(url, headers=headers, json=data, timeout=300)
|
||||
response.raise_for_status()
|
||||
result = response.json()
|
||||
|
||||
# Обработка разных форматов ответа (на всякий случай)
|
||||
if 'choices' in result and len(result['choices']) > 0:
|
||||
text = str(result['choices'][0]['message']['content'])
|
||||
return text, text
|
||||
else:
|
||||
return f"Неожиданный ответ от сервера: {result}", str(result)
|
||||
|
||||
except requests.exceptions.RequestException as e:
|
||||
return f"Ошибка при запросе к CustomProvider API: {e}", f"Ошибка: {e}"
|
||||
@@ -41,7 +41,7 @@ class GeminiProvider(BaseLLMProvider):
|
||||
|
||||
try:
|
||||
# 4. Отправляем POST-запрос с данными и настройками прокси
|
||||
response = requests.post(api_url, json=data, proxies=proxies, timeout=400)
|
||||
response = requests.post(api_url, json=data, proxies=proxies, timeout=90)
|
||||
|
||||
# Проверяем, не вернул ли сервер ошибку (например, 4xx или 5xx)
|
||||
response.raise_for_status()
|
||||
|
||||
@@ -3,13 +3,6 @@ from g4f.client import Client
|
||||
from .base_provider import BaseLLMProvider
|
||||
|
||||
class Gpt4FreeProvider(BaseLLMProvider):
|
||||
"""
|
||||
Провайдер для работы с моделью GPT через библиотеку gpt4free.
|
||||
|
||||
Этот класс реализует интерфейс BaseLLMProvider и предоставляет возможность
|
||||
взаимодействия с различными LLM через сервис gpt4free, который не требует
|
||||
API ключа для работы.
|
||||
"""
|
||||
# gpt4free не требует API ключа
|
||||
def __init__(self, api_key: str | None = None):
|
||||
super().__init__(api_key)
|
||||
@@ -17,18 +10,6 @@ class Gpt4FreeProvider(BaseLLMProvider):
|
||||
|
||||
|
||||
def generate(self, model: str, system_prompt: str, user_prompt: str, temp: float):
|
||||
"""
|
||||
Генерирует ответ от модели GPT с использованием gpt4free.
|
||||
|
||||
Args:
|
||||
model (str): Название модели для генерации ответа
|
||||
system_prompt (str): Системное сообщение для контекста
|
||||
user_prompt (str): Пользовательский запрос
|
||||
temp (float): Температура генерации ( controls randomness of responses)
|
||||
|
||||
Returns:
|
||||
tuple: Кортеж из двух одинаковых строк - сгенерированного ответа и его копии
|
||||
"""
|
||||
# temp в g4f может работать не для всех внутренних провайдеров
|
||||
try:
|
||||
response = self.client.chat.completions.create(
|
||||
|
||||
@@ -1,34 +1,23 @@
|
||||
import requests
|
||||
import openai
|
||||
from .base_provider import BaseLLMProvider
|
||||
|
||||
class IoNetProvider(BaseLLMProvider):
|
||||
def __init__(self, api_key: str):
|
||||
super().__init__(api_key)
|
||||
self.base_url = 'https://api.intelligence.io.solutions/api/v1'
|
||||
self.client = openai.OpenAI(
|
||||
api_key=self.api_key,
|
||||
base_url='https://api.intelligence.io.solutions/api/v1/'
|
||||
)
|
||||
|
||||
def generate(self, model: str, system_prompt: str, user_prompt: str, temp: float):
|
||||
url = f"{self.base_url}/chat/completions"
|
||||
headers = {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": f"Bearer {self.api_key}"
|
||||
}
|
||||
|
||||
data = {
|
||||
"model": model,
|
||||
"messages": [
|
||||
response = self.client.chat.completions.create(
|
||||
model=model,
|
||||
messages=[
|
||||
{'role': 'system', 'content': system_prompt},
|
||||
{'role': 'user', 'content': user_prompt},
|
||||
],
|
||||
"temperature": temp
|
||||
}
|
||||
|
||||
try:
|
||||
response = requests.post(url, headers=headers, json=data)
|
||||
response.raise_for_status()
|
||||
|
||||
result = response.json()
|
||||
text = str(result['choices'][0]['message']['content'])
|
||||
temperature=temp,
|
||||
stream=False
|
||||
)
|
||||
text = str(response.choices[0].message.content)
|
||||
return text, text # Возвращаем как чистый текст, так и Markdown
|
||||
|
||||
except requests.exceptions.RequestException as e:
|
||||
raise Exception(f"Ошибка при запросе к IO.net API: {e}")
|
||||
Reference in New Issue
Block a user