From f2c7dcbeb6076bbaf3f6258ec8452730732fc742 Mon Sep 17 00:00:00 2001 From: swrneko Date: Thu, 11 Sep 2025 14:04:21 +0300 Subject: [PATCH] add multiple audiofiles input and glue it --- app.py | 7 ++++--- config.py | 2 ++ handlers/fileHandlers.py | 18 +++++++++++++----- handlers/glueAudio.py | 11 +++++++++++ handlers/gradioHandler.py | 9 ++++++++- services/fasterWhisper.py | 2 +- 6 files changed, 39 insertions(+), 10 deletions(-) create mode 100644 handlers/glueAudio.py diff --git a/app.py b/app.py index 184ca37..e525b9a 100644 --- a/app.py +++ b/app.py @@ -11,8 +11,9 @@ from handlers.gradioHandler import GradioHandlers from handlers.fileHandlers import FileHandlers from services.fasterWhisper import FasterWhisper from handlers.convertMdToPdf import ConvertMdToPdf +from handlers.glueAudio import GlueAudio -gh = GradioHandlers(gr, Llm, ConvertMdToPdf, FileHandlers, FasterWhisper) +gh = GradioHandlers(gr, Llm, ConvertMdToPdf, FileHandlers, FasterWhisper, GlueAudio) def main(): with gr.Blocks() as demo: @@ -32,7 +33,7 @@ def main(): with gr.Row(): with gr.Accordion(label='Recognization and integration'): with gr.Column(): - audioFile = gr.Audio(label='Load audio for transcribe', type="filepath") + audioFiles = gr.Files(label='Load audio for transcribe', type="filepath", file_types=['audio']) images = gr.Files(label='Upload images', file_types=['image']) recognizeBtn = gr.Button('recognize and integrate', variant='primary') @@ -99,7 +100,7 @@ def main(): saveFileCheckbox.change(gh.updateTextbox, inputs=saveFileCheckbox, outputs=filename) saveFileCheckbox.change(gh.updateTextbox, inputs=saveFileCheckbox, outputs=filenamePdf) - recognizeBtn.click(gh.FasterWhisper.recognize, outputs=[recognizedText], inputs=[fastWhisperModel, device, compute_type, audioFile, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText]) + recognizeBtn.click(gh.handleRecognizeBtn, outputs=[recognizedText], inputs=[audioFiles, fastWhisperModel, device, compute_type, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText, gr.State(GLUED_AUDIO_FILENAME), gr.State(OUTPUT_PATH)]) # Если пайплайн включен то тогда делаем автоматически # автоматический пайплайн diff --git a/config.py b/config.py index 09c65e5..fdc64d6 100644 --- a/config.py +++ b/config.py @@ -13,6 +13,8 @@ DEFAULT_API_KEY=os.getenv('API_KEY') # Задаем выходную директорию OUTPUT_PATH='outputs' +GLUED_AUDIO_FILENAME='glued.mp3' + DEFAULT_SYSTEM_PROMPT='''You are a diligent university student who has recorded a lecture as an audio file and later transcribed it into raw text. Your task is to rewrite this unstructured transcript into a clear, logically organized, and detailed lecture summary (lecture notes). diff --git a/handlers/fileHandlers.py b/handlers/fileHandlers.py index 3fa15a0..d466100 100644 --- a/handlers/fileHandlers.py +++ b/handlers/fileHandlers.py @@ -1,11 +1,14 @@ from pathlib import Path -from markdown_pdf import MarkdownPdf # Для аннотации типов + +# Для аннотации типов +from markdown_pdf import MarkdownPdf +from pydub import AudioSegment class FileHandlers: # Функция сохранения файла - def saveFile(self, filename, content, output_path): + def saveFile(self, filename, content, output_path, format='mp3'): ''' - Сохраняет текст или pdf из markdown_pdf в файл с указанным названием и директорией. + Сохраняет текст, pdf из markdown_pdf или склеенный аудиофайл в файл с указанным названием и директорией. Args: :param filename: название файла; @@ -16,8 +19,13 @@ class FileHandlers: directory = Path(output_path) filePath = directory / filename # Добавление пути директории filePath.parent.mkdir(parents=True, exist_ok=True) # Создание директории если не существует + # Сохранение для разных типов if type(content) == MarkdownPdf: - content.save(filePath) + return content.save(filePath) + elif type(content) == str: - filePath.write_text(content, encoding='utf-8') + return filePath.write_text(content, encoding='utf-8') + + elif type(content) == AudioSegment: + return content.export(filePath, format=format) diff --git a/handlers/glueAudio.py b/handlers/glueAudio.py new file mode 100644 index 0000000..4c07a46 --- /dev/null +++ b/handlers/glueAudio.py @@ -0,0 +1,11 @@ +from pydub import AudioSegment + +class GlueAudio(): + def glue(self, audioFiles): + glued = AudioSegment.empty() + + for audioFile in audioFiles: + audio = AudioSegment.from_file(audioFile) + glued += audio + + return glued diff --git a/handlers/gradioHandler.py b/handlers/gradioHandler.py index b6923b5..f5286c5 100644 --- a/handlers/gradioHandler.py +++ b/handlers/gradioHandler.py @@ -1,12 +1,19 @@ class GradioHandlers: - def __init__(self, gr, Llm, ConvertMdToPdf, FileHandlers, FasterWhisper): + def __init__(self, gr, Llm, ConvertMdToPdf, FileHandlers, FasterWhisper, GlueAudio): # Объект для работы с файлами self.fh = FileHandlers() + self.ga = GlueAudio() self.ConvertMdToPdf = ConvertMdToPdf() self.FasterWhisper = FasterWhisper() self.Llm = Llm self.gr = gr + def handleRecognizeBtn(self, audioFiles, model, device, compute_type, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText, filename, outPath): + audioFile = self.ga.glue(audioFiles) + file = self.fh.saveFile(filename, audioFile, outPath) + + return self.FasterWhisper.recognize(model, device, compute_type, file, beamSize, vadFilter, minSilenceDurationMs, speechPadMs, temp0, temp1, temp2, wordTimestamps, noSpeechThreshold, conditionOnPreviousText) + # Функция улучшения текста def generateByCondition(self, api_key, llm_model, system_prompt, recognized_text, llm_temperature, is_pipeline_enabled, trigger, isSaveFile, filename, filenamePdf, output_path): llm = self.Llm(api_key) diff --git a/services/fasterWhisper.py b/services/fasterWhisper.py index 882468d..1d0349d 100644 --- a/services/fasterWhisper.py +++ b/services/fasterWhisper.py @@ -23,7 +23,7 @@ class FasterWhisper: for seg in segments: text += f"[{self.format_timestamp(seg.start)} -> {self.format_timestamp(seg.end)}] {seg.text}" + '\n' - return(text) + return text def format_timestamp(self, seconds: float) -> str: millis = int(seconds * 1000)