studio_V1_4_asr_GPT

Running

App Files Files Community

qqwjq1981 commited on Apr 11

Commit

8d5a056

verified ·

1 Parent(s): cc355be

Update app.py

Browse files

Files changed (1) hide show

app.py +51 -24

app.py CHANGED Viewed

@@ -125,32 +125,59 @@ def handle_feedback(feedback):
         return "Thank you for your feedback!", None
 def segment_background_audio(audio_path, background_audio_path="background_segments.wav"):
-    pipeline = Pipeline.from_pretrained("pyannote/voice-activity-detection", use_auth_token=hf_api_key)
-    vad_result = pipeline(audio_path)
-    full_audio = AudioSegment.from_wav(audio_path)
-    full_duration_sec = len(full_audio) / 1000.0
-    current_time = 0.0
-    result_audio = AudioSegment.empty()
-    for segment in vad_result.itersegments():
-        # Background segment before the speech
-        if current_time < segment.start:
-            bg = full_audio[int(current_time * 1000):int(segment.start * 1000)]
-            result_audio += bg
-        # Add silence for the speech duration
-        silence_duration = segment.end - segment.start
-        result_audio += AudioSegment.silent(duration=int(silence_duration * 1000))
-        current_time = segment.end
-    # Handle any remaining background after the last speech
-    if current_time < full_duration_sec:
-        result_audio += full_audio[int(current_time * 1000):]
-    result_audio.export(background_audio_path, format="wav")
     return background_audio_path
 def transcribe_video_with_speakers(video_path):
     # Extract audio from video
     video = VideoFileClip(video_path)

         return "Thank you for your feedback!", None
 def segment_background_audio(audio_path, background_audio_path="background_segments.wav"):
+    """
+    Uses Demucs to separate audio and extract background (non-vocal) parts.
+    Merges drums, bass, and other stems into a single background track.
+    """
+    # Step 1: Run Demucs using the 4-stem model
+    subprocess.run([
+        "demucs",
+        "-n", "htdemucs",  # 4-stem model
+        audio_path
+    ], check=True)
+    # Step 2: Locate separated stem files
+    filename = os.path.splitext(os.path.basename(audio_path))[0]
+    stem_dir = os.path.join("separated", "htdemucs", filename)
+    # Step 3: Load and merge background stems
+    drums = AudioSegment.from_wav(os.path.join(stem_dir, "drums.wav"))
+    bass = AudioSegment.from_wav(os.path.join(stem_dir, "bass.wav"))
+    other = AudioSegment.from_wav(os.path.join(stem_dir, "other.wav"))
+    background = drums.overlay(bass).overlay(other)
+    # Step 4: Export the merged background
+    background.export(background_audio_path, format="wav")
     return background_audio_path
+# def segment_background_audio(audio_path, background_audio_path="background_segments.wav"):
+#     pipeline = Pipeline.from_pretrained("pyannote/voice-activity-detection", use_auth_token=hf_api_key)
+#     vad_result = pipeline(audio_path)
+#     full_audio = AudioSegment.from_wav(audio_path)
+#     full_duration_sec = len(full_audio) / 1000.0
+#     current_time = 0.0
+#     result_audio = AudioSegment.empty()
+#     for segment in vad_result.itersegments():
+#         # Background segment before the speech
+#         if current_time < segment.start:
+#             bg = full_audio[int(current_time * 1000):int(segment.start * 1000)]
+#             result_audio += bg
+#         # Add silence for the speech duration
+#         silence_duration = segment.end - segment.start
+#         result_audio += AudioSegment.silent(duration=int(silence_duration * 1000))
+#         current_time = segment.end
+#     # Handle any remaining background after the last speech
+#     if current_time < full_duration_sec:
+#         result_audio += full_audio[int(current_time * 1000):]
+#     result_audio.export(background_audio_path, format="wav")
+#     return background_audio_path
 def transcribe_video_with_speakers(video_path):
     # Extract audio from video
     video = VideoFileClip(video_path)