Whisper_speaker_diarization2

Runtime error

App Files Files Community

raihanrifaldi commited on Dec 20, 2023

Commit

8bedf81

1 Parent(s): 7e18c2b

kembali ke file sebelumnya

Browse files

Files changed (1) hide show

app.py +147 -146

app.py CHANGED Viewed

@@ -29,105 +29,105 @@ import psutil
 whisper_models = ["tiny", "base", "small", "medium", "large-v1", "large-v2"]
 source_languages = {
-    "English": "English",
-    # "zh": "Chinese",
-    # "de": "German",
-    # "es": "Spanish",
-    # "ru": "Russian",
-    # "ko": "Korean",
-    # "fr": "French",
-    "Japan": "Japanese",
-    # "pt": "Portuguese",
-    # "tr": "Turkish",
-    # "pl": "Polish",
-    # "ca": "Catalan",
-    # "nl": "Dutch",
-    # "ar": "Arabic",
-    # "sv": "Swedish",
-    # "it": "Italian",
-    "Indonesia": "Indonesian",
-    # "hi": "Hindi",
-    # "fi": "Finnish",
-    # "vi": "Vietnamese",
-    # "he": "Hebrew",
-    # "uk": "Ukrainian",
-    # "el": "Greek",
-    "Malaysia": "Malay",
-    # "cs": "Czech",
-    # "ro": "Romanian",
-    # "da": "Danish",
-    # "hu": "Hungarian",
-    # "ta": "Tamil",
-    # "no": "Norwegian",
-    # "th": "Thai",
-    # "ur": "Urdu",
-    # "hr": "Croatian",
-    # "bg": "Bulgarian",
-    # "lt": "Lithuanian",
-    # "la": "Latin",
-    # "mi": "Maori",
-    # "ml": "Malayalam",
-    # "cy": "Welsh",
-    # "sk": "Slovak",
-    # "te": "Telugu",
-    # "fa": "Persian",
-    # "lv": "Latvian",
-    # "bn": "Bengali",
-    # "sr": "Serbian",
-    # "az": "Azerbaijani",
-    # "sl": "Slovenian",
-    # "kn": "Kannada",
-    # "et": "Estonian",
-    # "mk": "Macedonian",
-    # "br": "Breton",
-    # "eu": "Basque",
-    # "is": "Icelandic",
-    # "hy": "Armenian",
-    # "ne": "Nepali",
-    # "mn": "Mongolian",
-    # "bs": "Bosnian",
-    # "kk": "Kazakh",
-    # "sq": "Albanian",
-    # "sw": "Swahili",
-    # "gl": "Galician",
-    # "mr": "Marathi",
-    # "pa": "Punjabi",
-    # "si": "Sinhala",
-    # "km": "Khmer",
-    # "sn": "Shona",
-    # "yo": "Yoruba",
-    # "so": "Somali",
-    # "af": "Afrikaans",
-    # "oc": "Occitan",
-    # "ka": "Georgian",
-    # "be": "Belarusian",
-    # "tg": "Tajik",
-    # "sd": "Sindhi",
-    # "gu": "Gujarati",
-    # "am": "Amharic",
-    # "yi": "Yiddish",
-    # "lo": "Lao",
-    # "uz": "Uzbek",
-    # "fo": "Faroese",
-    # "ht": "Haitian creole",
-    # "ps": "Pashto",
-    # "tk": "Turkmen",
-    # "nn": "Nynorsk",
-    # "mt": "Maltese",
-    # "sa": "Sanskrit",
-    # "lb": "Luxembourgish",
-    # "my": "Myanmar",
-    # "bo": "Tibetan",
-    # "tl": "Tagalog",
-    # "mg": "Malagasy",
-    # "as": "Assamese",
-    # "tt": "Tatar",
-    # "haw": "Hawaiian",
-    # "ln": "Lingala",
-    # "ha": "Hausa",
-    # "ba": "Bashkir",
-    # "jw": "Javanese",
-    # "su": "Sundanese",
 }
 source_language_list = [key[0] for key in source_languages.items()]
@@ -227,7 +227,9 @@ def speech_to_text(video_file_path, selected_source_lang, whisper_model, num_spe
     Speaker diarization model and pipeline from by https://github.com/pyannote/pyannote-audio
     """
-    model = WhisperModel(whisper_model, device="cuda", compute_type="int8_float16")
     time_start = time.time()
     if(video_file_path == None):
         raise ValueError("Error no video input")
@@ -349,13 +351,13 @@ video_in = gr.Video(label="Video file", mirror_webcam=False)
 youtube_url_in = gr.Textbox(label="Youtube url", lines=1, interactive=True)
 df_init = pd.DataFrame(columns=['Start', 'End', 'Speaker', 'Text'])
 memory = psutil.virtual_memory()
-selected_source_lang = gr.Dropdown(choices=source_language_list, type="value", value="Indonesia", label="Bahasa yang digunakan dalam Video", interactive=True)
-selected_whisper_model = gr.Dropdown(choices=whisper_models, type="value", value="base", label="Whisper model", interactive=True)
-number_speakers = gr.Number(precision=0, value=0, label="Masukkan jumlah pembicara untuk hasil yang lebih baik. *Jika nilai=0, model akan otomatis menemukan jumlah pembicara terbaik", interactive=True)
 system_info = gr.Markdown(f"*Memory: {memory.total / (1024 * 1024 * 1024):.2f}GB, used: {memory.percent}%, available: {memory.available / (1024 * 1024 * 1024):.2f}GB*")
 download_transcript = gr.File(label="Download transcript")
 transcription_df = gr.DataFrame(value=df_init,label="Transcription dataframe", row_count=(0, "dynamic"), max_rows = 10, wrap=True, overflow_row_behaviour='paginate')
-title = "SPERCO"
 demo = gr.Blocks(title=title)
 demo.encrypt = False
@@ -364,19 +366,18 @@ with demo:
     with gr.Tab("Whisper speaker diarization"):
         gr.Markdown('''
             <div>
-            <h1 style='text-align: center'>SPERCO</h1>
-            Teknologi ini menggunakan Whisper models dari <a href='https://github.com/openai/whisper' target='_blank'><b>OpenAI</b></a> with <a href='https://github.com/guillaumekln/faster-whisper' target='_blank'><b>CTranslate2</b></a> yang merupakan mesin inferensi cepat untuk model Transformer untuk mengenali ucapan (4 kali lebih cepat dari model openai asli dengan akurasi yang sama)
-            dan model ECAPA-TDNN dari <a href='https://github.com/speechbrain/speechbrain' target='_blank'><b>SpeechBrain</b></a> untuk mengkodekan dan mengklasifikasikan pembicara.
             </div>
         ''')
         with gr.Row():
             gr.Markdown('''
             ### Transcribe youtube link using OpenAI Whisper
-            ##### 1. Menggunakan model Whisper dari OpenAI untuk memisahkan audio menjadi segmen dan menghasilkan transkripsi.
-            ##### 2. Menghasilkan pengkodan pembicara untuk setiap segmen.
-            ##### 3. Menerapkan pengelompokan aglomeratif pada pengkodean untuk mengidentifikasi pembicara untuk setiap segmen.
             ''')
         with gr.Row():
@@ -432,41 +433,41 @@ with demo:
-    # with gr.Tab("Whisper Transcribe Japanese Audio"):
-    #     gr.Markdown(f'''
-    #           <div>
-    #           <h1 style='text-align: center'>Whisper Transcribe Japanese Audio</h1>
-    #           </div>
-    #           Transcribe long-form microphone or audio inputs with the click of a button! The fine-tuned
-    #           checkpoint <a href='https://huggingface.co/{MODEL_NAME}' target='_blank'><b>{MODEL_NAME}</b></a> to transcribe audio files of arbitrary length.
-    #       ''')
-    #     microphone = gr.inputs.Audio(source="microphone", type="filepath", optional=True)
-    #     upload = gr.inputs.Audio(source="upload", type="filepath", optional=True)
-    #     transcribe_btn = gr.Button("Transcribe Audio")
-    #     text_output = gr.Textbox()
-    #     with gr.Row():
-    #         gr.Markdown('''
-    #             ### You can test by following examples:
-    #             ''')
-    #     examples = gr.Examples(examples=
-    #           [ "sample1.wav",
-    #             "sample2.wav",
-    #             ],
-    #           label="Examples", inputs=[upload])
-    #     transcribe_btn.click(transcribe, [microphone, upload], outputs=text_output)
-    # with gr.Tab("Whisper Transcribe Japanese YouTube"):
-    #     gr.Markdown(f'''
-    #           <div>
-    #           <h1 style='text-align: center'>Whisper Transcribe Japanese YouTube</h1>
-    #           </div>
-    #             Transcribe long-form YouTube videos with the click of a button! The fine-tuned checkpoint:
-    #             <a href='https://huggingface.co/{MODEL_NAME}' target='_blank'><b>{MODEL_NAME}</b></a> to transcribe audio files of arbitrary length.
-    #         ''')
-    #     youtube_link = gr.Textbox(label="Youtube url", lines=1, interactive=True)
-    #     yt_transcribe_btn = gr.Button("Transcribe YouTube")
-    #     text_output2 = gr.Textbox()
-    #     html_output = gr.Markdown()
-    #     yt_transcribe_btn.click(yt_transcribe, [youtube_link], outputs=[html_output, text_output2])
 demo.launch(debug=True)

 whisper_models = ["tiny", "base", "small", "medium", "large-v1", "large-v2"]
 source_languages = {
+    "en": "English",
+    "zh": "Chinese",
+    "de": "German",
+    "es": "Spanish",
+    "ru": "Russian",
+    "ko": "Korean",
+    "fr": "French",
+    "ja": "Japanese",
+    "pt": "Portuguese",
+    "tr": "Turkish",
+    "pl": "Polish",
+    "ca": "Catalan",
+    "nl": "Dutch",
+    "ar": "Arabic",
+    "sv": "Swedish",
+    "it": "Italian",
+    "id": "Indonesian",
+    "hi": "Hindi",
+    "fi": "Finnish",
+    "vi": "Vietnamese",
+    "he": "Hebrew",
+    "uk": "Ukrainian",
+    "el": "Greek",
+    "ms": "Malay",
+    "cs": "Czech",
+    "ro": "Romanian",
+    "da": "Danish",
+    "hu": "Hungarian",
+    "ta": "Tamil",
+    "no": "Norwegian",
+    "th": "Thai",
+    "ur": "Urdu",
+    "hr": "Croatian",
+    "bg": "Bulgarian",
+    "lt": "Lithuanian",
+    "la": "Latin",
+    "mi": "Maori",
+    "ml": "Malayalam",
+    "cy": "Welsh",
+    "sk": "Slovak",
+    "te": "Telugu",
+    "fa": "Persian",
+    "lv": "Latvian",
+    "bn": "Bengali",
+    "sr": "Serbian",
+    "az": "Azerbaijani",
+    "sl": "Slovenian",
+    "kn": "Kannada",
+    "et": "Estonian",
+    "mk": "Macedonian",
+    "br": "Breton",
+    "eu": "Basque",
+    "is": "Icelandic",
+    "hy": "Armenian",
+    "ne": "Nepali",
+    "mn": "Mongolian",
+    "bs": "Bosnian",
+    "kk": "Kazakh",
+    "sq": "Albanian",
+    "sw": "Swahili",
+    "gl": "Galician",
+    "mr": "Marathi",
+    "pa": "Punjabi",
+    "si": "Sinhala",
+    "km": "Khmer",
+    "sn": "Shona",
+    "yo": "Yoruba",
+    "so": "Somali",
+    "af": "Afrikaans",
+    "oc": "Occitan",
+    "ka": "Georgian",
+    "be": "Belarusian",
+    "tg": "Tajik",
+    "sd": "Sindhi",
+    "gu": "Gujarati",
+    "am": "Amharic",
+    "yi": "Yiddish",
+    "lo": "Lao",
+    "uz": "Uzbek",
+    "fo": "Faroese",
+    "ht": "Haitian creole",
+    "ps": "Pashto",
+    "tk": "Turkmen",
+    "nn": "Nynorsk",
+    "mt": "Maltese",
+    "sa": "Sanskrit",
+    "lb": "Luxembourgish",
+    "my": "Myanmar",
+    "bo": "Tibetan",
+    "tl": "Tagalog",
+    "mg": "Malagasy",
+    "as": "Assamese",
+    "tt": "Tatar",
+    "haw": "Hawaiian",
+    "ln": "Lingala",
+    "ha": "Hausa",
+    "ba": "Bashkir",
+    "jw": "Javanese",
+    "su": "Sundanese",
 }
 source_language_list = [key[0] for key in source_languages.items()]
     Speaker diarization model and pipeline from by https://github.com/pyannote/pyannote-audio
     """
+    # model = whisper.load_model(whisper_model)
+    # model = WhisperModel(whisper_model, device="cuda", compute_type="int8_float16")
+    model = WhisperModel(whisper_model, compute_type="int8")
     time_start = time.time()
     if(video_file_path == None):
         raise ValueError("Error no video input")
 youtube_url_in = gr.Textbox(label="Youtube url", lines=1, interactive=True)
 df_init = pd.DataFrame(columns=['Start', 'End', 'Speaker', 'Text'])
 memory = psutil.virtual_memory()
+selected_source_lang = gr.Dropdown(choices=source_language_list, type="value", value="en", label="Spoken language in video", interactive=True)
+selected_whisper_model = gr.Dropdown(choices=whisper_models, type="value", value="base", label="Selected Whisper model", interactive=True)
+number_speakers = gr.Number(precision=0, value=0, label="Input number of speakers for better results. If value=0, model will automatic find the best number of speakers", interactive=True)
 system_info = gr.Markdown(f"*Memory: {memory.total / (1024 * 1024 * 1024):.2f}GB, used: {memory.percent}%, available: {memory.available / (1024 * 1024 * 1024):.2f}GB*")
 download_transcript = gr.File(label="Download transcript")
 transcription_df = gr.DataFrame(value=df_init,label="Transcription dataframe", row_count=(0, "dynamic"), max_rows = 10, wrap=True, overflow_row_behaviour='paginate')
+title = "Whisper speaker diarization"
 demo = gr.Blocks(title=title)
 demo.encrypt = False
     with gr.Tab("Whisper speaker diarization"):
         gr.Markdown('''
             <div>
+            <h1 style='text-align: center'>Whisper speaker diarization</h1>
+            This space uses Whisper models from <a href='https://github.com/openai/whisper' target='_blank'><b>OpenAI</b></a> with <a href='https://github.com/guillaumekln/faster-whisper' target='_blank'><b>CTranslate2</b></a> which is a fast inference engine for Transformer models to recognize the speech (4 times faster than original openai model with same accuracy)
+            and ECAPA-TDNN model from <a href='https://github.com/speechbrain/speechbrain' target='_blank'><b>SpeechBrain</b></a> to encode and clasify speakers
             </div>
         ''')
         with gr.Row():
             gr.Markdown('''
             ### Transcribe youtube link using OpenAI Whisper
+            ##### 1. Using Open AI's Whisper model to seperate audio into segments and generate transcripts.
+            ##### 2. Generating speaker embeddings for each segments.
+            ##### 3. Applying agglomerative clustering on the embeddings to identify the speaker for each segment.
             ''')
         with gr.Row():
+    with gr.Tab("Whisper Transcribe Japanese Audio"):
+        gr.Markdown(f'''
+              <div>
+              <h1 style='text-align: center'>Whisper Transcribe Japanese Audio</h1>
+              </div>
+              Transcribe long-form microphone or audio inputs with the click of a button! The fine-tuned
+              checkpoint <a href='https://huggingface.co/{MODEL_NAME}' target='_blank'><b>{MODEL_NAME}</b></a> to transcribe audio files of arbitrary length.
+          ''')
+        microphone = gr.inputs.Audio(source="microphone", type="filepath", optional=True)
+        upload = gr.inputs.Audio(source="upload", type="filepath", optional=True)
+        transcribe_btn = gr.Button("Transcribe Audio")
+        text_output = gr.Textbox()
+        with gr.Row():
+            gr.Markdown('''
+                ### You can test by following examples:
+                ''')
+        examples = gr.Examples(examples=
+              [ "sample1.wav",
+                "sample2.wav",
+                ],
+              label="Examples", inputs=[upload])
+        transcribe_btn.click(transcribe, [microphone, upload], outputs=text_output)
+    with gr.Tab("Whisper Transcribe Japanese YouTube"):
+        gr.Markdown(f'''
+              <div>
+              <h1 style='text-align: center'>Whisper Transcribe Japanese YouTube</h1>
+              </div>
+                Transcribe long-form YouTube videos with the click of a button! The fine-tuned checkpoint:
+                <a href='https://huggingface.co/{MODEL_NAME}' target='_blank'><b>{MODEL_NAME}</b></a> to transcribe audio files of arbitrary length.
+            ''')
+        youtube_link = gr.Textbox(label="Youtube url", lines=1, interactive=True)
+        yt_transcribe_btn = gr.Button("Transcribe YouTube")
+        text_output2 = gr.Textbox()
+        html_output = gr.Markdown()
+        yt_transcribe_btn.click(yt_transcribe, [youtube_link], outputs=[html_output, text_output2])
 demo.launch(debug=True)