add multiple audiofiles input and glue it

This commit is contained in:
swrneko
2025-09-11 14:04:21 +03:00
parent 6b27149460
commit f2c7dcbeb6
6 changed files with 39 additions and 10 deletions

7
app.py
View File

@@ -11,8 +11,9 @@ from handlers.gradioHandler import GradioHandlers
from handlers.fileHandlers import FileHandlers
from services.fasterWhisper import FasterWhisper
from handlers.convertMdToPdf import ConvertMdToPdf
from handlers.glueAudio import GlueAudio
gh = GradioHandlers(gr, Llm, ConvertMdToPdf, FileHandlers, FasterWhisper)
gh = GradioHandlers(gr, Llm, ConvertMdToPdf, FileHandlers, FasterWhisper, GlueAudio)
def main():
with gr.Blocks() as demo:
@@ -32,7 +33,7 @@ def main():
with gr.Row():
with gr.Accordion(label='Recognization and integration'):
with gr.Column():
audioFile = gr.Audio(label='Load audio for transcribe', type="filepath")
audioFiles = gr.Files(label='Load audio for transcribe', type="filepath", file_types=['audio'])
images = gr.Files(label='Upload images', file_types=['image'])
recognizeBtn = gr.Button('recognize and integrate', variant='primary')
@@ -99,7 +100,7 @@ def main():
saveFileCheckbox.change(gh.updateTextbox, inputs=saveFileCheckbox, outputs=filename)
saveFileCheckbox.change(gh.updateTextbox, inputs=saveFileCheckbox, outputs=filenamePdf)
recognizeBtn.click(gh.FasterWhisper.recognize, outputs=[recognizedText], inputs=[fastWhisperModel, device, compute_type, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText])
recognizeBtn.click(gh.handleRecognizeBtn, outputs=[recognizedText], inputs=[audioFiles, fastWhisperModel, device, compute_type, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText, gr.State(GLUED_AUDIO_FILENAME), gr.State(OUTPUT_PATH)])
# Если пайплайн включен то тогда делаем автоматически
# автоматический пайплайн

View File

@@ -13,6 +13,8 @@ DEFAULT_API_KEY=os.getenv('API_KEY')
# Задаем выходную директорию
OUTPUT_PATH='outputs'
GLUED_AUDIO_FILENAME='glued.mp3'
DEFAULT_SYSTEM_PROMPT='''You are a diligent university student who has recorded a lecture as an audio file and later transcribed it into raw text.
Your task is to rewrite this unstructured transcript into a clear, logically organized, and detailed lecture summary (lecture notes).

View File

@@ -1,11 +1,14 @@
from pathlib import Path
from markdown_pdf import MarkdownPdf # Для аннотации типов
# Для аннотации типов
from markdown_pdf import MarkdownPdf
from pydub import AudioSegment
class FileHandlers:
# Функция сохранения файла
def saveFile(self, filename, content, output_path):
def saveFile(self, filename, content, output_path, format='mp3'):
'''
Сохраняет текст или pdf из markdown_pdf в файл с указанным названием и директорией.
Сохраняет текст, pdf из markdown_pdf или склеенный аудиофайл в файл с указанным названием и директорией.
Args:
:param filename: название файла;
@@ -16,8 +19,13 @@ class FileHandlers:
directory = Path(output_path)
filePath = directory / filename # Добавление пути директории
filePath.parent.mkdir(parents=True, exist_ok=True) # Создание директории если не существует
# Сохранение для разных типов
if type(content) == MarkdownPdf:
content.save(filePath)
return content.save(filePath)
elif type(content) == str:
filePath.write_text(content, encoding='utf-8')
return filePath.write_text(content, encoding='utf-8')
elif type(content) == AudioSegment:
return content.export(filePath, format=format)

11
handlers/glueAudio.py Normal file
View File

@@ -0,0 +1,11 @@
from pydub import AudioSegment
class GlueAudio():
def glue(self, audioFiles):
glued = AudioSegment.empty()
for audioFile in audioFiles:
audio = AudioSegment.from_file(audioFile)
glued += audio
return glued

View File

@@ -1,12 +1,19 @@
class GradioHandlers:
def __init__(self, gr, Llm, ConvertMdToPdf, FileHandlers, FasterWhisper):
def __init__(self, gr, Llm, ConvertMdToPdf, FileHandlers, FasterWhisper, GlueAudio):
# Объект для работы с файлами
self.fh = FileHandlers()
self.ga = GlueAudio()
self.ConvertMdToPdf = ConvertMdToPdf()
self.FasterWhisper = FasterWhisper()
self.Llm = Llm
self.gr = gr
def handleRecognizeBtn(self, audioFiles, model, device, compute_type, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText, filename, outPath):
audioFile = self.ga.glue(audioFiles)
file = self.fh.saveFile(filename, audioFile, outPath)
return self.FasterWhisper.recognize(model, device, compute_type, file, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText)
# Функция улучшения текста
def generateByCondition(self, api_key, llm_model, system_prompt, recognized_text, llm_temperature, is_pipeline_enabled, trigger, isSaveFile, filename, filenamePdf, output_path):
llm = self.Llm(api_key)

View File

@@ -23,7 +23,7 @@ class FasterWhisper:
for seg in segments:
text += f"[{self.format_timestamp(seg.start)} -> {self.format_timestamp(seg.end)}] {seg.text}" + '\n'
return(text)
return text
def format_timestamp(self, seconds: float) -> str:
millis = int(seconds * 1000)