#!/usr/bin/env python3
"""VoiceGrab — cut a time range from a video file and export a clean MP3 voice sample.
- Pick an input video (mp4, mkv, webm, avi, mov, ...)
- Set start / stop timestamps
- Optional: isolate voice with noise reduction (no GPU or big models needed)
- Export MP3 (320 kbps, loudness-normalized to -16 LUFS — good for AI voice training)
- Waveform preview with the selected range highlighted
"""
from __future__ import annotations
import os
import platform
import shutil
import subprocess
import sys
import tempfile
import wave
APP_NAME = "VoiceGrab"
APP_VERSION = "1.0.0"
# ---------------------------------------------------------------------------
# Resource / executable helpers
# ---------------------------------------------------------------------------
def app_dir() -> str:
"""Directory that contains our files (bundled data dir in frozen builds)."""
if getattr(sys, "frozen", False):
return sys._MEIPASS # type: ignore[attr-defined]
return os.path.dirname(os.path.abspath(__file__))
def appdata_dir() -> str:
"""Per-user directory for saved model noise profiles."""
if platform.system() == "Windows":
base = os.environ.get("APPDATA", os.path.expanduser("~"))
elif platform.system() == "Darwin":
base = os.path.expanduser("~/Library/Application Support")
else:
base = os.environ.get("XDG_DATA_HOME", os.path.expanduser("~/.local/share"))
d = os.path.join(base, APP_NAME)
os.makedirs(d, exist_ok=True)
return d
# ---------------------------------------------------------------------------
# ffmpeg helpers (bundled binary in frozen builds, system ffmpeg otherwise)
# ---------------------------------------------------------------------------
def ffmpeg_exe() -> str:
if getattr(sys, "frozen", False):
exe = os.path.join(app_dir(), "ffmpeg.exe" if platform.system() == "Windows" else "ffmpeg")
if os.path.exists(exe):
# ensure executable bit
try:
os.chmod(exe, 0o755)
except OSError:
pass
return exe
return "ffmpeg"
def ffprobe_exe() -> str:
if getattr(sys, "frozen", False):
exe = os.path.join(app_dir(), "ffprobe.exe" if platform.system() == "Windows" else "ffprobe")
if os.path.exists(exe):
try:
os.chmod(exe, 0o755)
except OSError:
pass
return exe
return "ffprobe"
class FfmpegError(RuntimeError):
pass
def probe_duration(path: str) -> float:
try:
out = subprocess.run(
[ffprobe_exe(), "-hide_banner", "-v", "error",
"-show_entries", "format=duration", "-of", "default=nw=1:nk=1", path],
capture_output=True, text=True, timeout=60,
)
return float(out.stdout.strip().splitlines()[0])
except (ValueError, IndexError, subprocess.SubprocessError) as exc:
raise FfmpegError(f"Could not read media file: {path}") from exc
def extract_wav(path: str, start: float, stop: float, workdir: str) -> str:
"""Extract the selected range as mono 48 kHz WAV (PCM 16-bit)."""
out = os.path.join(workdir, "clip.wav")
cmd = [
ffmpeg_exe(), "-hide_banner", "-v", "error", "-y",
"-ss", f"{start:.3f}", "-to", f"{stop:.3f}",
"-i", path,
"-vn", "-ac", "1", "-ar", "48000",
"-c:a", "pcm_s16le", out,
]
r = subprocess.run(cmd, capture_output=True, text=True, timeout=600)
if r.returncode != 0 or not os.path.exists(out):
raise FfmpegError(f"Audio extraction failed:\n{r.stderr.strip()[:1500]}")
return out
def wav_stats(path: str):
"""Return (nframes, samplerate, rms) of a 16-bit WAV."""
with wave.open(path, "rb") as w:
n = w.getnframes()
sr = w.getframerate()
data = w.readframes(n)
import array
a = array.array("h")
a.frombytes(data[: len(data) // 2 * 2])
if len(a) == 0:
return n, sr, 0.0
rms = (sum((x / 32768.0) ** 2 for x in a) / len(a)) ** 0.5
return n, sr, float(rms)
def peaks(path: str, buckets: int = 2000):
"""Downsample a 16-bit WAV to (max, min) per bucket for plotting."""
with wave.open(path, "rb") as w:
n = w.getnframes()
data = w.readframes(n)
import array
a = array.array("h")
a.frombytes(data[: len(data) // 2 * 2])
m = len(a)
if m == 0:
return [], []
step = max(1, m // buckets)
maxs, mins = [], []
for i in range(0, m, step):
chunk = a[i:i + step]
maxs.append(max(chunk) / 32768.0)
mins.append(min(chunk) / 32768.0)
return maxs, mins
# ---------------------------------------------------------------------------
# Voice isolation (lightweight, CPU-only)
# ---------------------------------------------------------------------------
try:
import numpy as np
import noisereduce as nr
HAVE_NR = True
NR_IMPORT_ERROR = ""
except Exception as _exc: # pragma: no cover
HAVE_NR = False
NR_IMPORT_ERROR = str(_exc)
class VoiceReducer:
"""STFT noise reduction with an optional noise reference clip."""
def __init__(self, intensity: float, stationary: bool):
self.intensity = float(intensity) # 0.0 .. 1.0
self.stationary = stationary
self.noise_ref = None # optional numpy array of noise
def set_noise_ref(self, noise):
self.noise_ref = noise
def reduce(self, y: "np.ndarray", sr: int) -> "np.ndarray":
kwargs = dict(
y=y, sr=sr,
prop_decrease=min(0.95, self.intensity),
stationary=self.stationary,
use_tqdm=False,
)
if self.noise_ref is not None:
kwargs["y_noise"] = self.noise_ref
return nr.reduce_noise(**kwargs)
# ---------------------------------------------------------------------------
# Audio IO helpers (numpy/soundfile)
# ---------------------------------------------------------------------------
def read_wav_f32(path: str):
import soundfile as sf
data, sr = sf.read(path, dtype="float32")
return data, sr
def write_wav_f32(path: str, y, sr: int):
import soundfile as sf
sf.write(path, y, sr, subtype="PCM_16")
# ---------------------------------------------------------------------------
# GUI
# ---------------------------------------------------------------------------
def _import_pyside():
from PySide6 import QtCore, QtGui, QtWidgets # noqa: F401
return QtCore, QtGui, QtWidgets
def run_gui() -> int:
QtCore, QtGui, QtWidgets = _import_pyside()
def _normalize(y):
import numpy as _np
p = _np.percentile(_np.abs(y), 99.5)
if p < 1e-6:
return y
return (y / p * 0.891) # ~ -1 dBFS reference peak
class Worker(QtCore.QThread):
log = QtCore.Signal(str)
done = QtCore.Signal(object, str) # (success, message)
def __init__(self, job):
super().__init__()
self.job = job
def run(self):
import numpy as _np
try:
self.log.emit("Extracting audio range…")
with tempfile.TemporaryDirectory(prefix="voicegrab-") as wd:
wav = extract_wav(self.job["input"], self.job["start"], self.job["stop"], wd)
y, sr = read_wav_f32(wav)
if self.job["isolate"]:
if not HAVE_NR:
raise FfmpegError(f"Voice isolation unavailable: {NR_IMPORT_ERROR}")
self.log.emit("Isolating voice (noise reduction)\u2026")
red = VoiceReducer(self.job["intensity"], self.job["stationary"])
y = red.reduce(y, sr)
else:
self.log.emit("Normalizing loudness…")
y = _normalize(y)
work_wav = os.path.join(wd, "work.wav")
write_wav_f32(work_wav, y, sr)
self.log.emit("Encoding MP3 (320 kbps, loudness-normalized)…")
enc = [
ffmpeg_exe(), "-hide_banner", "-v", "error", "-y",
"-i", work_wav,
"-af", f"loudnorm=I={self.job['lufts']}:TP=-1.5:LRA=11",
"-c:a", "libmp3lame", "-b:a", "320k", self.job["output"],
]
r = subprocess.run(enc, capture_output=True, text=True, timeout=300)
if r.returncode != 0 or not os.path.exists(self.job["output"]):
raise FfmpegError(f"MP3 encoding failed:\n{r.stderr.strip()[:1500]}")
self.log.emit("Done.")
self.done.emit(True, self.job["output"])
except Exception as exc:
self.done.emit(False, str(exc))
return _main_loop(QtCore, QtGui, QtWidgets, Worker, _normalize)
def _main_loop(QtCore, QtGui, QtWidgets, Worker, _normalize) -> int:
app = QtWidgets.QApplication(sys.argv)
app.setApplicationName(APP_NAME)
app.setApplicationVersion(APP_VERSION)
class MainWindow(QtWidgets.QMainWindow):
def __init__(self):
super().__init__()
self.setWindowTitle(f"{APP_NAME} — MP3 voice sample cutter")
self.resize(920, 640)
self.input_path = None
self.duration = 0.0
self.worker = None
self._build()
# ---------- UI ----------
def _build(self):
central = QtWidgets.QWidget()
self.setCentralWidget(central)
lay = QtWidgets.QVBoxLayout(central)
lay.setContentsMargins(14, 14, 14, 14)
lay.setSpacing(12)
# Row 1: input file
row1 = QtWidgets.QHBoxLayout()
row1.addWidget(QtWidgets.QLabel("Video / audio file:"))
self.in_edit = QtWidgets.QLineEdit()
self.in_edit.setPlaceholderText("Choose an .mp4 / .mkv / .mov / .webm … file")
row1.addWidget(self.in_edit, 1)
btn_open = QtWidgets.QPushButton("Open…")
btn_open.clicked.connect(self.choose_input)
row1.addWidget(btn_open)
self.lbl_info = QtWidgets.QLabel("")
self.lbl_info.setStyleSheet("color:#64748b;")
row1.addWidget(self.lbl_info)
lay.addLayout(row1)
# Row 2: times
row2 = QtWidgets.QHBoxLayout()
row2.addWidget(QtWidgets.QLabel("Start:"))
self.t_start = QtWidgets.QLineEdit("0:00.00")
self.t_start.setFixedWidth(110)
row2.addWidget(self.t_start)
row2.addSpacing(12)
row2.addWidget(QtWidgets.QLabel("Stop:"))
self.t_stop = QtWidgets.QLineEdit("0:00.00")
self.t_stop.setFixedWidth(110)
row2.addWidget(self.t_stop)
self.lbl_range = QtWidgets.QLabel("")
self.lbl_range.setStyleSheet("color:#64748b;")
row2.addWidget(self.lbl_range)
row2.addStretch(1)
btn_preview = QtWidgets.QPushButton("Load preview")
btn_preview.setToolTip("Extract the selected range so you can listen before saving")
btn_preview.clicked.connect(self.load_preview)
row2.addWidget(btn_preview)
btn_listen = QtWidgets.QPushButton("▶ Play range")
btn_listen.clicked.connect(self.play_range)
row2.addWidget(btn_listen)
lay.addLayout(row2)
# Row 3: isolation + output
self.chk_isolate = QtWidgets.QCheckBox("Isolate voice (reduce background noise)")
self.chk_isolate.setChecked(True)
self.chk_isolate.toggled.connect(self._isolate_toggled)
lay.addWidget(self.chk_isolate)
iso_row = QtWidgets.QHBoxLayout()
iso_row.addWidget(QtWidgets.QLabel("Noise reduction strength:"))
self.spn_intensity = QtWidgets.QSpinBox()
self.spn_intensity.setRange(0, 100)
self.spn_intensity.setValue(80)
self.spn_intensity.setSuffix("%")
iso_row.addWidget(self.spn_intensity)
iso_row.addSpacing(16)
self.chk_stationary = QtWidgets.QCheckBox(
"Stationary noise (fan / hum — better if constant)")
iso_row.addWidget(self.chk_stationary)
iso_row.addStretch(1)
self.lbl_iso_status = QtWidgets.QLabel("")
self.lbl_iso_status.setStyleSheet("color:#64748b;")
iso_row.addWidget(self.lbl_iso_status)
lay.addLayout(iso_row)
self.iso_row_widget = iso_row
out_row = QtWidgets.QHBoxLayout()
out_row.addWidget(QtWidgets.QLabel("Output MP3:"))
self.out_edit = QtWidgets.QLineEdit()
self.out_edit.setPlaceholderText("Defaults to _voicegrab.mp3 next to the source")
out_row.addWidget(self.out_edit, 1)
btn_out = QtWidgets.QPushButton("Browse…")
btn_out.clicked.connect(self.choose_output)
out_row.addWidget(btn_out)
lay.addLayout(out_row)
# Waveform
lay.addWidget(QtWidgets.QLabel("Waveform of the selected range:"))
self.wave = WaveView()
self.wave.setMinimumHeight(160)
lay.addWidget(self.wave, 1)
# Bottom bar
bar = QtWidgets.QHBoxLayout()
self.btn_export = QtWidgets.QPushButton("Export MP3")
self.btn_export.setMinimumHeight(40)
self.btn_export.setStyleSheet(
"font-size:14px; font-weight:600; background:#0ea5e9; color:white;"
"border-radius:6px; padding:4px 18px;")
self.btn_export.clicked.connect(self.export)
bar.addWidget(self.btn_export)
self.lbl_status = QtWidgets.QLabel("Ready.")
self.lbl_status.setStyleSheet("color:#64748b;")
bar.addWidget(self.lbl_status, 1)
lay.addLayout(bar)
if not HAVE_NR:
self.lbl_iso_status.setText(f"⚠ voice isolation unavailable ({NR_IMPORT_ERROR})")
self.chk_isolate.setEnabled(False)
# ---------- helpers ----------
def _isolate_toggled(self, on):
for w in self.iso_row_widget.items():
if isinstance(w, QtWidgets.QWidget) and w not in (self.lbl_iso_status,):
w.setEnabled(on)
@staticmethod
def _parse_time(s: str) -> float:
"""Accept 125, 1:05, 1:05.5, 1:05:00, 01:00:00.500."""
s = s.strip()
parts = s.split(":")
if len(parts) > 3:
raise ValueError("Invalid time format")
t = 0.0
for i, p in enumerate(parts):
v = float(p)
t = t * 60 + v
return max(0.0, t)
def _fmt_time(self, t: float) -> str:
m, s = divmod(t, 60)
h, m = divmod(int(m), 60)
if h:
return f"{h}:{int(m):02d}:{s:05.2f}"
return f"{int(m)}:{s:05.2f}"
def _times(self):
start = self._parse_time(self.t_start.text() or "0")
stop = self._parse_time(self.t_stop.text() or "0")
if start >= stop:
raise ValueError("Start must be before stop.")
return start, stop
def refresh_info(self):
if self.input_path and self.duration:
self.lbl_info.setText(
f"Duration {self._fmt_time(self.duration)} Range {self._range_text()}")
else:
self.lbl_info.setText("")
def _range_text(self):
try:
s, e = self._times()
return f"{s:.2f}s – {e:.2f}s (clip {e - s:.2f}s)"
except ValueError:
return "— invalid range —"
# ---------- slots ----------
def choose_input(self):
path, _ = QtWidgets.QFileDialog.getOpenFileName(
self, "Choose video or audio file", "",
"Media files (*.mp4 *.mkv *.mov *.webm *.avi *.m4a *.mp3 *.wav *.ogg *.flv *.ts);;All files (*)")
if not path:
return
self.input_path = path
self.in_edit.setText(path)
self.t_stop.setText(self._fmt_time(self.duration)) # after probe below
try:
self.duration = probe_duration(path)
self.t_stop.setText(self._fmt_time(self.duration))
except FfmpegError as exc:
self.status(str(exc)); return
self.refresh_info()
self.load_preview()
def choose_output(self):
if not self.input_path:
return
base = os.path.splitext(os.path.basename(self.input_path))[0]
start = os.path.join(os.path.dirname(self.input_path), base + "_voicegrab.mp3")
path, _ = QtWidgets.QFileDialog.getSaveFileName(
self, "Save MP3 as", start, "MP3 files (*.mp3)")
if path:
self.out_edit.setText(path)
def load_preview(self):
if not self.input_path:
return
try:
s, e = self._times()
except ValueError as exc:
self.status(str(exc)); return
try:
with tempfile.TemporaryDirectory(prefix="voicegrab-") as wd:
wav = extract_wav(self.input_path, s, e, wd)
maxs, mins = peaks(wav)
n, sr, rms = wav_stats(wav)
except FfmpegError as exc:
self.status(str(exc)); return
self.wave.set_peaks(maxs, mins)
self.lbl_range.setText(f"clip {n / sr:.2f}s @ {sr} Hz RMS {rms:.3f}")
def play_range(self):
if not self.input_path:
return
try:
s, e = self._times()
except ValueError as exc:
self.status(str(exc)); return
player = os.path.join(tempfile.gettempdir(), "voicegrab_preview.mp3")
try:
cmd = [
ffmpeg_exe(), "-hide_banner", "-v", "error", "-y",
"-ss", f"{s:.3f}", "-to", f"{e:.3f}", "-i", self.input_path,
"-vn", "-c:a", "libmp3lame", "-b:a", "128k", player,
]
subprocess.run(cmd, check=True, capture_output=True, text=True, timeout=300)
if platform.system() == "Windows":
os.startfile(player) # type: ignore[attr-defined]
elif platform.system() == "Darwin":
subprocess.Popen(["open", player])
else:
subprocess.Popen(["xdg-open", player])
self.status("Playing range in system player…")
except (subprocess.SubprocessError, OSError) as exc:
self.status(f"Could not play range: {exc}")
def status(self, msg: str):
self.lbl_status.setText(msg)
def export(self):
if self.worker is not None:
self.status("Already working…"); return
if not self.input_path:
QtWidgets.QMessageBox.warning(self, APP_NAME, "Choose an input file first.")
return
try:
s, e = self._times()
except ValueError as exc:
QtWidgets.QMessageBox.warning(self, APP_NAME, str(exc)); return
out = self.out_edit.text().strip()
if not out:
base = os.path.splitext(os.path.basename(self.input_path))[0]
out = os.path.join(os.path.dirname(os.path.abspath(self.input_path)),
base + "_voicegrab.mp3")
self.btn_export.setEnabled(False)
self.status("Working… (see log in status area)")
self.worker = Worker({
"input": self.input_path, "start": s, "stop": e,
"output": out,
"isolate": self.chk_isolate.isChecked(),
"intensity": self.spn_intensity.value() / 100.0,
"stationary": self.chk_stationary.isChecked(),
"lufts": -16,
})
self.worker.log.connect(self.status)
self.worker.done.connect(self._export_done)
self.worker.start()
def _export_done(self, ok, msg):
self.btn_export.setEnabled(True)
self.worker = None
if ok:
self.status(f"✔ Saved: {msg}")
QtWidgets.QMessageBox.information(
self, APP_NAME,
f"MP3 saved to:\n{msg}\n\nTip: 30–120 s of clean speech is usually "
f"enough for high-quality AI voice cloning.")
else:
self.status(f"✖ {msg.splitlines()[0]}")
QtWidgets.QMessageBox.critical(self, APP_NAME, msg)
# ---------------- wave view ----------------
class WaveView(QtWidgets.QWidget):
def __init__(self):
super().__init__()
self.maxs: list = []
self.mins: list = []
self.setMinimumSize(120, 80)
def set_peaks(self, maxs, mins):
self.maxs = list(maxs)
self.mins = list(mins)
self.update()
def paintEvent(self, ev):
with QtGui.QPainter(self) as p:
w, h = self.width(), self.height()
mid = h / 2
p.fillRect(0, 0, w, h, QtGui.QColor("#0f172a"))
p.setPen(QtGui.QColor("#334155"))
p.drawLine(0, int(mid), w, int(mid))
if not self.maxs:
p.setPen(QtGui.QColor("#94a3b8"))
p.drawText(0, 0, w, h, int(QtCore.Qt.AlignCenter), "Open a file to see the waveform")
return
n = len(self.maxs)
color = QtGui.QColor("#38bdf8")
for i, (mx, mn) in enumerate(zip(self.maxs, self.mins)):
x = i * w / n
y1 = mid - mx * (mid - 4)
y2 = mid - mn * (mid - 4)
p.setPen(color)
p.drawLine(int(x), int(y1), int(x), max(int(y1), int(y2)))
win = MainWindow()
win.show()
return app.exec()
if __name__ == "__main__":
if "--cli" in sys.argv:
# Simple headless mode: voicegrab --cli [output.mp3]
sys.argv = [a for a in sys.argv if a != "--cli"]
if len(sys.argv) < 4:
print("usage: voicegrab --cli [output.mp3]")
raise SystemExit(2)
inp, s, e = sys.argv[1], float(sys.argv[2]), float(sys.argv[3])
out = sys.argv[4] if len(sys.argv) > 4 else os.path.join(
os.path.dirname(os.path.abspath(inp)),
os.path.splitext(os.path.basename(inp))[0] + "_voicegrab.mp3")
import numpy as _np
with tempfile.TemporaryDirectory(prefix="voicegrab-") as wd:
wav = extract_wav(inp, s, e, wd)
y, sr = read_wav_f32(wav)
if HAVE_NR:
red = VoiceReducer(0.8, stationary=True)
y = red.reduce(y, sr)
red_wav = os.path.join(wd, "reduced.wav")
write_wav_f32(red_wav, y, sr)
subprocess.run([
ffmpeg_exe(), "-hide_banner", "-v", "error", "-y",
"-i", red_wav, "-af", "loudnorm=I=-16:TP=-1.5:LRA=11",
"-c:a", "libmp3lame", "-b:a", "320k", out,
], check=True)
print(f"Saved {out}")
else:
raise SystemExit(run_gui())