#!/usr/bin/env python3 """ Singing Segment Detector for Live Stream Recordings ===================================================== Detects singing segments in long stream recordings and exports DaVinci Resolve-compatible EDL markers for fast video editing. Pipeline: 1. Extract audio from video (ffmpeg → 16kHz mono WAV) 2. Run inaSpeechSegmenter to classify speech/music/noise 3. Merge nearby "music" segments (singing) with configurable gap 4. Export to EDL (Edit Decision List) for DaVinci Resolve import 5. Optionally export CSV for review Usage: python detect_singing.py "path/to/stream_recording.mp4" python detect_singing.py "path/to/stream_recording.mp4" --gap 15 --min-duration 30 python detect_singing.py "path/to/stream_recording.mp4" --output-dir "D:/singing_edits" """ import argparse import csv import json import os import subprocess import sys import tempfile import time from pathlib import Path # ────────────────────────────────────────────── # Step 1: Extract audio from video using ffmpeg # ────────────────────────────────────────────── def extract_audio(video_path: str, output_wav: str, ffmpeg_bin: str = "ffmpeg") -> str: """Extract audio from video file as 16kHz mono WAV (required by inaSpeechSegmenter).""" print(f"\n[Step 1/4] Extracting audio from: {video_path}") print(f" Output WAV: {output_wav}") cmd = [ ffmpeg_bin, "-i", video_path, "-vn", # no video "-acodec", "pcm_s16le", # 16-bit PCM "-ar", "16000", # 16kHz sample rate "-ac", "1", # mono "-y", # overwrite output_wav, ] try: result = subprocess.run( cmd, capture_output=True, text=True, timeout=1800, # 30 min timeout for very long files ) if result.returncode != 0: print(f" [ERROR] ffmpeg failed:\n{result.stderr[-500:]}") sys.exit(1) except FileNotFoundError: print(f" [ERROR] ffmpeg not found at '{ffmpeg_bin}'.") print(f" Make sure ffmpeg is installed and in your PATH.") sys.exit(1) size_mb = os.path.getsize(output_wav) / (1024 * 1024) print(f" Audio extracted: {size_mb:.1f} MB") return output_wav # ────────────────────────────────────────────── # Step 2: Run inaSpeechSegmenter # ────────────────────────────────────────────── def run_segmentation(wav_path: str) -> list: """ Run inaSpeechSegmenter on audio file. Returns list of (label, start_sec, end_sec) tuples. Labels: 'music', 'speech', 'male', 'female', 'noise', 'noEnergy' NOTE: inaSpeechSegmenter classifies singing voice as 'music'. """ print(f"\n[Step 2/4] Running audio segmentation (this may take a while)...") try: from inaSpeechSegmenter import Segmenter except ImportError: print(" [ERROR] inaSpeechSegmenter is not installed.") print(" Install it with: pip install inaSpeechSegmenter") sys.exit(1) start_time = time.time() # vad_engine='smn' → speech/music/noise detection # detect_gender=False → faster, we don't need gender info seg = Segmenter(vad_engine='smn', detect_gender=False) segments = seg(wav_path) elapsed = time.time() - start_time print(f" Segmentation complete in {elapsed:.1f}s") print(f" Total segments found: {len(segments)}") # Count segment types type_counts = {} type_durations = {} for label, start, end in segments: type_counts[label] = type_counts.get(label, 0) + 1 type_durations[label] = type_durations.get(label, 0) + (end - start) for label in sorted(type_counts.keys()): count = type_counts[label] dur = type_durations[label] print(f" {label:>10s}: {count:4d} segments, {format_timecode(dur)} total") return segments # ────────────────────────────────────────────── # Step 3: Filter & merge singing (music) segments # ────────────────────────────────────────────── def merge_singing_segments( segments: list, max_gap: float = 10.0, min_duration: float = 20.0, padding: float = 2.0, ) -> list: """ Filter for 'music' segments (which includes singing) and merge nearby segments that are likely the same song. Args: segments: Raw segmentation output [(label, start, end), ...] max_gap: Max gap (seconds) between music segments to merge them. Stream singers often have brief pauses, audience interaction between verses, etc. Default 10s works well. min_duration: Minimum duration (seconds) for a merged segment to be kept. Filters out short music stings, sound effects, etc. A typical song is 2-5 minutes, so 20s is a safe minimum. padding: Seconds to add before/after each segment for clean cuts. Returns: List of dicts: [{"start": float, "end": float, "duration": float, "index": int}, ...] """ print(f"\n[Step 3/4] Merging singing segments (gap={max_gap}s, min={min_duration}s, pad={padding}s)") # Extract only music segments music_segs = [(start, end) for label, start, end in segments if label == "music"] if not music_segs: print(" No music segments found!") return [] # Sort by start time (should already be sorted, but just in case) music_segs.sort(key=lambda x: x[0]) # Merge segments with gaps smaller than max_gap merged = [] current_start, current_end = music_segs[0] for start, end in music_segs[1:]: if start - current_end <= max_gap: # Extend current segment current_end = max(current_end, end) else: # Save current segment and start new one merged.append((current_start, current_end)) current_start, current_end = start, end merged.append((current_start, current_end)) # Apply padding and minimum duration filter results = [] idx = 1 for start, end in merged: duration = end - start if duration >= min_duration: padded_start = max(0, start - padding) padded_end = end + padding results.append({ "index": idx, "start": padded_start, "end": padded_end, "duration": padded_end - padded_start, "original_start": start, "original_end": end, }) idx += 1 print(f" Found {len(results)} singing segments after merge+filter:") for seg in results: print(f" Song {seg['index']:2d}: " f"{format_timecode(seg['start'])} → {format_timecode(seg['end'])} " f"({seg['duration']:.0f}s)") return results # ────────────────────────────────────────────── # Step 4: Export formats # ────────────────────────────────────────────── def export_edl(segments: list, output_path: str, fps: float = 29.97, title: str = "Singing Segments", source_filename: str = None): """ Export an EDL (Edit Decision List) file for DaVinci Resolve. DaVinci Resolve import: File → Import → Timeline → select the .edl file. The EDL creates cut points at each singing segment's in/out points. Args: source_filename: Original video filename (e.g. "stream_2026-04-01.mp4"). Used as reel name + FROM CLIP NAME so DaVinci Resolve matches the EDL to the correct clip in the Media Pool. Without this, Resolve picks a random clip when multiple videos are loaded in the same project. """ print(f"\n[Step 4/4] Exporting EDL: {output_path}") # DaVinci Resolve uses both the reel name column AND the "* FROM CLIP NAME:" # comment to identify which media clip an EDL edit refers to. # # Reel name: traditionally 8 chars max, but Resolve accepts longer names. # We use the filename without extension, truncated to a safe length. # FROM CLIP NAME: Resolve's preferred matching method. Must be the exact # filename (with extension) as it appears in the Media Pool. if source_filename: clip_name = source_filename # full filename with extension reel_name = Path(source_filename).stem[:32] # stem, truncated for safety # Replace spaces with underscores in reel name (some EDL parsers choke on spaces) reel_name_safe = reel_name.replace(" ", "_") else: clip_name = None reel_name_safe = "001" # Calculate record timecodes: sequential placement on the output timeline # Each segment is placed one after another, starting at 01:00:00:00 rec_offset = 3600.0 # start at 01:00:00:00 with open(output_path, "w", encoding="utf-8") as f: f.write(f"TITLE: {title}\n") f.write("FCM: NON-DROP FRAME\n\n") current_rec_pos = rec_offset for seg in segments: edit_num = f"{seg['index']:03d}" src_in = seconds_to_timecode(seg["start"], fps) src_out = seconds_to_timecode(seg["end"], fps) rec_in = seconds_to_timecode(current_rec_pos, fps) rec_out = seconds_to_timecode(current_rec_pos + seg["duration"], fps) # EDL format: edit# reel_name track_type transition src_in src_out rec_in rec_out f.write(f"{edit_num} {reel_name_safe} V C {src_in} {src_out} {rec_in} {rec_out}\n") # "* FROM CLIP NAME:" is the key line DaVinci uses for media matching. # It must exactly match the clip name shown in the Media Pool. if clip_name: f.write(f"* FROM CLIP NAME: {clip_name}\n") f.write(f"* COMMENT: Song {seg['index']} - Duration {seg['duration']:.0f}s\n\n") current_rec_pos += seg["duration"] print(f" EDL file written with {len(segments)} edits") if clip_name: print(f" Source clip name: {clip_name}") def export_csv(segments: list, output_path: str): """Export singing segments as CSV for review or further processing.""" with open(output_path, "w", newline="", encoding="utf-8") as f: writer = csv.writer(f) writer.writerow(["index", "start_sec", "end_sec", "duration_sec", "start_timecode", "end_timecode"]) for seg in segments: writer.writerow([ seg["index"], f"{seg['start']:.2f}", f"{seg['end']:.2f}", f"{seg['duration']:.1f}", format_timecode(seg["start"]), format_timecode(seg["end"]), ]) print(f" CSV file written: {output_path}") def export_davinci_markers_csv(segments: list, output_path: str, fps: float = 29.97): """ Export a CSV file that can be used with DaVinci Resolve's 'Import Markers from CSV' function (Resolve 18+). Format: #, Color, Name, Start TC, End TC, Duration TC, Notes """ with open(output_path, "w", newline="", encoding="utf-8") as f: writer = csv.writer(f) # DaVinci Resolve Marker CSV header writer.writerow(["#", "Color", "Name", "Start TC", "End TC", "Duration TC", "Notes"]) for seg in segments: start_tc = seconds_to_timecode(seg["start"], fps) end_tc = seconds_to_timecode(seg["end"], fps) dur_tc = seconds_to_timecode(seg["duration"], fps) writer.writerow([ seg["index"], "Blue", f"Song {seg['index']}", start_tc, end_tc, dur_tc, f"Duration: {seg['duration']:.0f}s", ]) print(f" DaVinci markers CSV written: {output_path}") def export_json(segments: list, output_path: str): """Export as JSON for programmatic use.""" with open(output_path, "w", encoding="utf-8") as f: json.dump(segments, f, indent=2, ensure_ascii=False) print(f" JSON file written: {output_path}") def export_ffmpeg_concat(segments: list, video_path: str, output_path: str): """ Export a .bat/.sh script that uses ffmpeg to directly cut and concatenate all singing segments into a single video file. This is the fully automated option — no DaVinci needed. """ video_name = Path(video_path).stem ext = Path(video_path).suffix out_video = str(Path(output_path).parent / f"{video_name}_singing_only{ext}") is_windows = os.name == "nt" or sys.platform == "win32" script_ext = ".bat" if is_windows else ".sh" script_path = str(Path(output_path).with_suffix(script_ext)) lines = [] if is_windows: lines.append("@echo off") lines.append("REM Auto-generated ffmpeg script to extract singing segments") lines.append(f'REM Source: {video_path}') lines.append("") else: lines.append("#!/bin/bash") lines.append("# Auto-generated ffmpeg script to extract singing segments") lines.append(f'# Source: {video_path}') lines.append("") # Create a concat file list concat_list_path = str(Path(output_path).parent / f"{video_name}_concat_list.txt") # Generate individual segment extraction commands segment_files = [] for seg in segments: seg_file = f"_seg_{seg['index']:03d}{ext}" seg_path = str(Path(output_path).parent / seg_file) segment_files.append(seg_file) start = seg["start"] duration = seg["duration"] cmd = (f'ffmpeg -y -ss {start:.2f} -i "{video_path}" ' f'-t {duration:.2f} -c copy "{seg_path}"') lines.append(f"echo Extracting Song {seg['index']}...") lines.append(cmd) lines.append("") # Generate concat list file lines.append(f"echo Creating concat list...") if is_windows: lines.append(f'(') for sf in segment_files: seg_path = str(Path(output_path).parent / sf) lines.append(f' echo file \'{seg_path}\'') lines.append(f') > "{concat_list_path}"') else: for i, sf in enumerate(segment_files): seg_path = str(Path(output_path).parent / sf) op = ">" if i == 0 else ">>" lines.append(f"echo \"file '{seg_path}'\" {op} \"{concat_list_path}\"") lines.append("") lines.append(f"echo Concatenating all segments...") lines.append(f'ffmpeg -y -f concat -safe 0 -i "{concat_list_path}" -c copy "{out_video}"') lines.append("") # Cleanup temp segment files lines.append("echo Cleaning up temp files...") for sf in segment_files: seg_path = str(Path(output_path).parent / sf) if is_windows: lines.append(f'del "{seg_path}"') else: lines.append(f'rm -f "{seg_path}"') if is_windows: lines.append(f'del "{concat_list_path}"') else: lines.append(f'rm -f "{concat_list_path}"') lines.append("") lines.append(f'echo Done! Output: {out_video}') if is_windows: lines.append("pause") with open(script_path, "w", encoding="utf-8") as f: f.write("\n".join(lines)) if not is_windows: os.chmod(script_path, 0o755) print(f" FFmpeg concat script written: {script_path}") print(f" Run it to auto-generate: {out_video}") # ────────────────────────────────────────────── # Export raw segmentation for debugging # ────────────────────────────────────────────── def export_raw_segments(segments: list, output_path: str): """Export ALL raw segments (speech/music/noise) as CSV for debugging.""" with open(output_path, "w", newline="", encoding="utf-8") as f: writer = csv.writer(f) writer.writerow(["label", "start_sec", "end_sec", "duration_sec", "start_timecode", "end_timecode"]) for label, start, end in segments: writer.writerow([ label, f"{start:.2f}", f"{end:.2f}", f"{end - start:.1f}", format_timecode(start), format_timecode(end), ]) print(f" Raw segments CSV written: {output_path}") # ────────────────────────────────────────────── # Utility functions # ────────────────────────────────────────────── def format_timecode(seconds: float) -> str: """Format seconds as HH:MM:SS.""" h = int(seconds // 3600) m = int((seconds % 3600) // 60) s = int(seconds % 60) return f"{h:02d}:{m:02d}:{s:02d}" def seconds_to_timecode(seconds: float, fps: float = 29.97) -> str: """Convert seconds to SMPTE timecode HH:MM:SS:FF.""" total_frames = int(seconds * fps) ff = total_frames % int(round(fps)) total_seconds = total_frames // int(round(fps)) ss = total_seconds % 60 total_minutes = total_seconds // 60 mm = total_minutes % 60 hh = total_minutes // 60 return f"{hh:02d}:{mm:02d}:{ss:02d}:{ff:02d}" # ────────────────────────────────────────────── # Main # ────────────────────────────────────────────── def main(): parser = argparse.ArgumentParser( description="Detect singing segments in stream recordings and export DaVinci Resolve markers.", formatter_class=argparse.RawDescriptionHelpFormatter, epilog=""" Examples: python detect_singing.py "D:\\streams\\2026-04-04.mp4" python detect_singing.py "D:\\streams\\2026-04-04.mp4" --gap 15 --min-duration 30 python detect_singing.py "D:\\streams\\2026-04-04.mp4" --output-dir "D:\\singing_edits" python detect_singing.py "D:\\streams\\2026-04-04.mp4" --auto-cut Tips: --gap 10 Merge music segments with <=10s gap (default). Increase if the singer often pauses >10s between verses in the same song. --min-duration 20 Ignore segments shorter than 20s (default). Decrease if the singer does very short songs or covers. --padding 2 Add 2s padding before/after each segment (default). Helps catch the very start/end of songs. --auto-cut Generate an ffmpeg script to automatically cut and concat all singing segments into a single video file. --fps 29.97 Frame rate for timecode calculation (default: 29.97 for NTSC). Use 25 for PAL or 23.976 for film. """, ) parser.add_argument("video", help="Path to the stream recording video file") parser.add_argument("--output-dir", "-o", default=None, help="Output directory (default: same directory as input video)") parser.add_argument("--gap", "-g", type=float, default=10.0, help="Max gap (seconds) to merge nearby singing segments (default: 10)") parser.add_argument("--min-duration", "-m", type=float, default=20.0, help="Minimum singing segment duration in seconds (default: 20)") parser.add_argument("--padding", "-p", type=float, default=2.0, help="Padding (seconds) before/after each segment (default: 2)") parser.add_argument("--fps", type=float, default=29.97, help="Video frame rate for timecode export (default: 29.97)") parser.add_argument("--keep-wav", action="store_true", help="Keep the extracted WAV file (default: delete after processing)") parser.add_argument("--auto-cut", action="store_true", help="Generate an ffmpeg script to auto-cut and concat singing segments") parser.add_argument("--raw-segments", action="store_true", help="Also export all raw segments (speech/music/noise) as CSV") parser.add_argument("--ffmpeg", default="ffmpeg", help="Path to ffmpeg binary (default: 'ffmpeg' from PATH)") args = parser.parse_args() # Validate input video_path = os.path.abspath(args.video) if not os.path.isfile(video_path): print(f"[ERROR] Video file not found: {video_path}") sys.exit(1) # Set output directory if args.output_dir: output_dir = os.path.abspath(args.output_dir) os.makedirs(output_dir, exist_ok=True) else: output_dir = os.path.dirname(video_path) video_stem = Path(video_path).stem print("=" * 60) print(" Singing Segment Detector") print("=" * 60) print(f" Input: {video_path}") print(f" Output dir: {output_dir}") print(f" Settings: gap={args.gap}s, min={args.min_duration}s, pad={args.padding}s") print(f" FPS: {args.fps}") # Step 1: Extract audio wav_path = os.path.join(output_dir, f"{video_stem}_audio.wav") extract_audio(video_path, wav_path, ffmpeg_bin=args.ffmpeg) # Step 2: Run segmentation raw_segments = run_segmentation(wav_path) # Step 3: Merge singing segments singing_segments = merge_singing_segments( raw_segments, max_gap=args.gap, min_duration=args.min_duration, padding=args.padding, ) if not singing_segments: print("\n[RESULT] No singing segments detected. Try lowering --min-duration or increasing --gap.") # Cleanup WAV if not args.keep_wav and os.path.exists(wav_path): os.remove(wav_path) sys.exit(0) # Step 4: Export results edl_path = os.path.join(output_dir, f"{video_stem}_singing.edl") csv_path = os.path.join(output_dir, f"{video_stem}_singing.csv") markers_path = os.path.join(output_dir, f"{video_stem}_markers.csv") json_path = os.path.join(output_dir, f"{video_stem}_singing.json") export_edl(singing_segments, edl_path, fps=args.fps, title=f"{video_stem} - Singing", source_filename=Path(video_path).name) export_csv(singing_segments, csv_path) export_davinci_markers_csv(singing_segments, markers_path, fps=args.fps) export_json(singing_segments, json_path) if args.auto_cut: export_ffmpeg_concat(singing_segments, video_path, json_path) if args.raw_segments: raw_csv_path = os.path.join(output_dir, f"{video_stem}_raw_segments.csv") export_raw_segments(raw_segments, raw_csv_path) # Cleanup WAV unless --keep-wav if not args.keep_wav and os.path.exists(wav_path): os.remove(wav_path) print(f"\n Temp WAV file deleted.") # Summary total_singing = sum(s["duration"] for s in singing_segments) print("\n" + "=" * 60) print(" DONE!") print("=" * 60) print(f" Songs detected: {len(singing_segments)}") print(f" Total singing time: {format_timecode(total_singing)}") print(f" EDL file: {edl_path}") print(f" DaVinci markers CSV: {markers_path}") print(f" Segments CSV: {csv_path}") print(f" Segments JSON: {json_path}") if args.auto_cut: script_ext = ".bat" if (os.name == "nt" or sys.platform == "win32") else ".sh" print(f" Auto-cut script: {json_path.replace('.json', script_ext)}") print() print(" To import into DaVinci Resolve:") print(" Option A: File → Import → Timeline → select the .edl file") print(" Option B: Timeline menu → Import Markers from CSV → select _markers.csv") print("=" * 60) if __name__ == "__main__": main()