v1: svd prototype — single-file singing detector
HuggingFace matthijs/svd (torch.jit) model + ffmpeg audio extract + tkinter file picker. Proves the concept in one script. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
7
.gitignore
vendored
Normal file
7
.gitignore
vendored
Normal file
@@ -0,0 +1,7 @@
|
|||||||
|
__pycache__/
|
||||||
|
*.pyc
|
||||||
|
*.wav
|
||||||
|
*.mp4
|
||||||
|
.DS_Store
|
||||||
|
venv/
|
||||||
|
.venv/
|
||||||
108
sing_detect.py
Normal file
108
sing_detect.py
Normal file
@@ -0,0 +1,108 @@
|
|||||||
|
import subprocess
|
||||||
|
import os
|
||||||
|
import torch
|
||||||
|
import librosa
|
||||||
|
import numpy as np
|
||||||
|
from huggingface_hub import hf_hub_download
|
||||||
|
import tkinter as tk
|
||||||
|
from tkinter import filedialog
|
||||||
|
|
||||||
|
# ------------------------
|
||||||
|
# 1. Extract audio from video
|
||||||
|
# ------------------------
|
||||||
|
def extract_audio(input_video, output_wav):
|
||||||
|
cmd = [
|
||||||
|
"ffmpeg", "-y",
|
||||||
|
"-i", input_video,
|
||||||
|
"-vn", # no video
|
||||||
|
"-ac", "1", # mono
|
||||||
|
"-ar", "16000", # 16kHz
|
||||||
|
output_wav
|
||||||
|
]
|
||||||
|
subprocess.run(cmd, check=True)
|
||||||
|
|
||||||
|
# ------------------------
|
||||||
|
# 2. Load pretrained model
|
||||||
|
# ------------------------
|
||||||
|
def load_model():
|
||||||
|
repo_id = "matthijs/svd" # HuggingFace singing voice detection model
|
||||||
|
filename = hf_hub_download(repo_id=repo_id, filename="model.pt")
|
||||||
|
model = torch.jit.load(filename)
|
||||||
|
model.eval()
|
||||||
|
return model
|
||||||
|
|
||||||
|
# ------------------------
|
||||||
|
# 3. Run inference on audio
|
||||||
|
# ------------------------
|
||||||
|
def detect_singing(audio_path, model, hop_length=512, threshold=0.5, min_duration=10.0):
|
||||||
|
y, sr = librosa.load(audio_path, sr=16000)
|
||||||
|
x = torch.tensor(y).float().unsqueeze(0)
|
||||||
|
|
||||||
|
with torch.no_grad():
|
||||||
|
pred = model(x).squeeze().numpy()
|
||||||
|
|
||||||
|
times = librosa.frames_to_time(np.arange(len(pred)), sr=sr, hop_length=hop_length)
|
||||||
|
|
||||||
|
segments = []
|
||||||
|
in_segment = False
|
||||||
|
seg_start = None
|
||||||
|
|
||||||
|
for t, p in zip(times, pred):
|
||||||
|
if p > threshold and not in_segment:
|
||||||
|
in_segment = True
|
||||||
|
seg_start = t
|
||||||
|
elif p <= threshold and in_segment:
|
||||||
|
in_segment = False
|
||||||
|
seg_end = t
|
||||||
|
if seg_end - seg_start >= min_duration:
|
||||||
|
segments.append((seg_start, seg_end))
|
||||||
|
|
||||||
|
if in_segment:
|
||||||
|
seg_end = times[-1]
|
||||||
|
if seg_end - seg_start >= min_duration:
|
||||||
|
segments.append((seg_start, seg_end))
|
||||||
|
|
||||||
|
return segments
|
||||||
|
|
||||||
|
# ------------------------
|
||||||
|
# 4. Export CSV
|
||||||
|
# ------------------------
|
||||||
|
def export_csv(segments, out_file="segments.csv"):
|
||||||
|
with open(out_file, "w", encoding="utf-8") as f:
|
||||||
|
f.write("start_time,end_time\n")
|
||||||
|
for s, e in segments:
|
||||||
|
f.write(f"{s:.2f},{e:.2f}\n")
|
||||||
|
print(f"[+] Saved: {out_file}")
|
||||||
|
|
||||||
|
# ------------------------
|
||||||
|
# MAIN
|
||||||
|
# ------------------------
|
||||||
|
if __name__ == "__main__":
|
||||||
|
# Tkinter GUI to choose file
|
||||||
|
root = tk.Tk()
|
||||||
|
root.withdraw()
|
||||||
|
input_video = filedialog.askopenfilename(
|
||||||
|
title="Choose your video file",
|
||||||
|
filetypes=[("Video files", "*.mp4 *.mkv *.flv *.mov *.avi"), ("All files", "*.*")]
|
||||||
|
)
|
||||||
|
if not input_video:
|
||||||
|
print("No file selected. Exiting.")
|
||||||
|
exit()
|
||||||
|
|
||||||
|
audio_file = "audio.wav"
|
||||||
|
|
||||||
|
print(f"[+] Processing: {input_video}")
|
||||||
|
|
||||||
|
# Step 1: extract audio
|
||||||
|
extract_audio(input_video, audio_file)
|
||||||
|
|
||||||
|
# Step 2: load model
|
||||||
|
model = load_model()
|
||||||
|
|
||||||
|
# Step 3: run detection
|
||||||
|
segments = detect_singing(audio_file, model)
|
||||||
|
|
||||||
|
# Step 4: save results
|
||||||
|
export_csv(segments, "segments.csv")
|
||||||
|
|
||||||
|
print("[+] Done! Detected singing segments have been saved to segments.csv")
|
||||||
Reference in New Issue
Block a user