From 1a529ff95bb66157c8cf6032f4cddb3b8eaab2d0 Mon Sep 17 00:00:00 2001 From: Joel Jose Date: Sat, 21 Mar 2026 16:43:18 +0530 Subject: [PATCH 1/6] Remove denoising code and add versioning Remove denoise_spectral(), denoise_morphological(), --denoise, and --denoise-input CLI flags. Denoising is available separately via the audio_denoising repo. Add VERSION file (2.0.0), __version__, and --version CLI flag. Make -i/--input required (no longer optional due to --denoise-input mode). Fixes #3 --- VERSION | 1 + visualmic.py | 112 +++++++++------------------------------------------ 2 files changed, 21 insertions(+), 92 deletions(-) create mode 100644 VERSION diff --git a/VERSION b/VERSION new file mode 100644 index 0000000..227cea2 --- /dev/null +++ b/VERSION @@ -0,0 +1 @@ +2.0.0 diff --git a/visualmic.py b/visualmic.py index f05ac8e..bee71ce 100644 --- a/visualmic.py +++ b/visualmic.py @@ -1,12 +1,24 @@ +""" +Visual Microphone: Recover sound from video using 2D DTCWT. + +Recovers sound from high-speed video by analyzing sub-pixel surface vibrations. +Uses the phase of complex wavelet coefficients to detect motion far too small +to see with the naked eye, then reconstructs an audible signal. + +Based on: Davis et al., "The Visual Microphone: Passive Recovery of Sound +from Video", ACM Transactions on Graphics (SIGGRAPH 2014). +""" + +__version__ = "2.0.0" + import argparse import os import sys import time + from scipy import signal -from scipy import ndimage import numpy as np import cv2 -from scipy.io.wavfile import read as read_wav from scipy.io.wavfile import write @@ -34,58 +46,6 @@ def save_wav(samples, output_name, sample_rate): print(f"Output saved to {output_name}") -def denoise_spectral(samples, fs, noise_duration=0.1): - f, t, Zxx = signal.stft(samples, fs=fs, nperseg=512) - magnitude = np.abs(Zxx) - phase = np.angle(Zxx) - - # Estimate noise from first noise_duration seconds - noise_frames = max(1, int(noise_duration * fs / (512 // 4))) - noise_profile = np.mean(magnitude[:, :noise_frames], axis=1, keepdims=True) - - # Subtract noise, floor at zero - clean_mag = np.maximum(magnitude - noise_profile, 0.0) - - # Reconstruct with original phase - clean_Zxx = clean_mag * np.exp(1j * phase) - _, reconstructed = signal.istft(clean_Zxx, fs=fs, nperseg=512) - - # Normalize to [-1, 1] - peak = np.max(np.abs(reconstructed)) - if peak > 0: - reconstructed = reconstructed / peak - return reconstructed - - -def denoise_morphological(samples, fs, threshold=20, amp=10): - f, t, Zxx = signal.stft(samples, fs=fs, nperseg=512) - magnitude = np.abs(Zxx) - - # Convert to grayscale (0-255) - mag_max = np.max(magnitude) - if mag_max == 0: - return samples - gray = magnitude * (255.0 / mag_max) - - # Binary threshold - mask = gray >= threshold - - # Morphological erosion then dilation - mask = ndimage.binary_erosion(mask, iterations=1) - mask = ndimage.binary_dilation(mask, iterations=2) - - # Apply mask: amplify signal, attenuate noise - masked_Zxx = np.where(mask, Zxx * amp, Zxx / amp) - - _, reconstructed = signal.istft(masked_Zxx, fs=fs, nperseg=512) - - # Normalize to [-1, 1] - peak = np.max(np.abs(reconstructed)) - if peak > 0: - reconstructed = reconstructed / peak - return reconstructed - - def postprocess_phase_signals(phase_signals, frame_count, nlevels, n_orient, ref_level, ref_orient, fps, freq_low=None, freq_high=None): # Temporal bandpass filtering nyquist = fps / 2.0 @@ -304,46 +264,22 @@ def extract_audio_gpu(cap, frame_count, nlevels, n_orient, ref_index, ref_orient def main(): parser = argparse.ArgumentParser(description='Visual Microphone: Recover sound from video using 2D DTCWT') - parser.add_argument('-i', '--input', default=None, help='Specify input video path') - parser.add_argument('-o', '--output', default='sound.wav', help='Specify output audio path (default: sound.wav)') + parser.add_argument( + '--version', action='version', + version=f'%(prog)s {__version__}' + ) + parser.add_argument('-i', '--input', required=True, help='Input video path') + parser.add_argument('-o', '--output', default='sound.wav', help='Output audio path (default: sound.wav)') parser.add_argument('-fl', '--freq-low', type=float, default=None, help='Lower cutoff frequency in Hz for temporal bandpass filter') parser.add_argument('-fh', '--freq-high', type=float, default=None, help='Upper cutoff frequency in Hz for temporal bandpass filter') parser.add_argument('--fps', type=float, default=None, help='Override video frame rate (Hz) for audio output sample rate') parser.add_argument('--roi', type=str, default=None, help='Region of interest as x,y,w,h (e.g. --roi 100,50,200,150)') parser.add_argument('--gpu', action='store_true', help='Use GPU-accelerated DTCWT (requires CUDA and pytorch_wavelets)') parser.add_argument('--batch-size', type=int, default=16, help='Frames per GPU batch (default: 16, GPU mode only)') - parser.add_argument('--denoise', choices=['spectral', 'morphological'], default=None, help='Audio denoising method (applied after reconstruction)') - parser.add_argument('--denoise-input', type=str, default=None, help='Denoise an existing WAV file instead of processing video') args = parser.parse_args() pipeline_start = time.time() - # Standalone denoise mode - if args.denoise_input is not None: - if args.denoise is None: - print("Error: --denoise-input requires --denoise {spectral,morphological}") - sys.exit(1) - if not os.path.isfile(args.denoise_input): - print(f"Error: file '{args.denoise_input}' not found") - sys.exit(1) - sr, wav_data = read_wav(args.denoise_input) - samples = wav_data.astype(np.float64) / 32767.0 - print(f"Loaded {args.denoise_input}: {len(samples)} samples, {sr} Hz, {len(samples)/sr:.2f}s") - print(f"Applying {args.denoise} denoising...") - if args.denoise == 'spectral': - samples = denoise_spectral(samples, sr) - else: - samples = denoise_morphological(samples, sr) - print("Denoising complete") - save_wav(samples, args.output, sr) - print(f"Total time: {format_duration(time.time() - pipeline_start)}") - return - - # Full pipeline mode — require -i - if args.input is None: - print("Error: -i/--input is required (or use --denoise-input for standalone denoising)") - sys.exit(1) - filename = args.input output_name = args.output freq_low = args.freq_low @@ -445,14 +381,6 @@ def main(): else: sound_data = extract_audio(cap, frame_count, nlevels, n_orient, ref_index, ref_orient, ref_level, fps, freq_low, freq_high, roi) - if args.denoise: - print(f"Applying {args.denoise} denoising...") - if args.denoise == 'spectral': - sound_data = denoise_spectral(sound_data, int(fps)) - else: - sound_data = denoise_morphological(sound_data, int(fps)) - print("Denoising complete") - save_wav(sound_data, output_name, int(fps)) print(f"Total time: {format_duration(time.time() - pipeline_start)}") From ed21b4069bb8c3108919b786f23881fba0c15971 Mon Sep 17 00:00:00 2001 From: Joel Jose Date: Sat, 21 Mar 2026 16:44:37 +0530 Subject: [PATCH 2/6] Add --nlevels, --biort, --qshift flags and GPU memory estimation Make wavelet parameters configurable: --nlevels (default 3), --biort (default near_sym_b), --qshift (default qshift_b). Defaults match DTCWT Motion Mag v2.0.0. Both CPU and GPU paths now use the same filter defaults for consistency. Add estimate_vram() and pre-flight VRAM check before GPU processing. Warns if estimated usage exceeds 70% of available VRAM with actionable suggestions. Fixes #4 --- visualmic.py | 54 +++++++++++++++++++++++++++++++++++++++++++++------- 1 file changed, 47 insertions(+), 7 deletions(-) diff --git a/visualmic.py b/visualmic.py index bee71ce..66f2778 100644 --- a/visualmic.py +++ b/visualmic.py @@ -99,9 +99,9 @@ def postprocess_phase_signals(phase_signals, frame_count, nlevels, n_orient, ref return sound_data -def extract_audio(cap, frame_count, nlevels, n_orient, ref_index, ref_orient, ref_level, fps, freq_low=None, freq_high=None, roi=None): +def extract_audio(cap, frame_count, nlevels, n_orient, ref_index, ref_orient, ref_level, fps, freq_low=None, freq_high=None, roi=None, biort='near_sym_b', qshift='qshift_b'): import dtcwt - transform = dtcwt.Transform2d() + transform = dtcwt.Transform2d(biort=biort, qshift=qshift) ref_conj = None phase_signals = [] progress_interval = max(1, frame_count // 10) @@ -154,12 +154,25 @@ def extract_audio(cap, frame_count, nlevels, n_orient, ref_index, ref_orient, re return postprocess_phase_signals(phase_signals, frame_count, nlevels, n_orient, ref_level, ref_orient, fps, freq_low, freq_high) -def extract_audio_gpu(cap, frame_count, nlevels, n_orient, ref_index, ref_orient, ref_level, fps, freq_low=None, freq_high=None, roi=None, batch_size=16): +def estimate_vram(batch_size, height, width, nlevels): + """Estimate peak GPU VRAM usage in bytes. + + Peak occurs during batched forward DTCWT: input frames plus + transform intermediates (~15x overhead per frame). + """ + frame_bytes = height * width * 4 # float32 + dtcwt_overhead = 15 # empirical: forward transform intermediates + batch_vram = batch_size * frame_bytes * dtcwt_overhead + pytorch_overhead = 300 * 1024 * 1024 # ~300 MB for PyTorch + filter weights + return batch_vram + pytorch_overhead + + +def extract_audio_gpu(cap, frame_count, nlevels, n_orient, ref_index, ref_orient, ref_level, fps, freq_low=None, freq_high=None, roi=None, batch_size=16, biort='near_sym_b', qshift='qshift_b'): import torch from pytorch_wavelets import DTCWTForward device = torch.device('cuda') - xfm = DTCWTForward(J=nlevels, biort='near_sym_b', qshift='qshift_b').to(device) + xfm = DTCWTForward(J=nlevels, biort=biort, qshift=qshift).to(device) print(f"GPU mode: {torch.cuda.get_device_name(0)}, batch_size={batch_size}") ref_coeffs = None @@ -276,6 +289,9 @@ def main(): parser.add_argument('--roi', type=str, default=None, help='Region of interest as x,y,w,h (e.g. --roi 100,50,200,150)') parser.add_argument('--gpu', action='store_true', help='Use GPU-accelerated DTCWT (requires CUDA and pytorch_wavelets)') parser.add_argument('--batch-size', type=int, default=16, help='Frames per GPU batch (default: 16, GPU mode only)') + parser.add_argument('--nlevels', type=int, default=3, help='Number of DTCWT decomposition levels (default: 3)') + parser.add_argument('--biort', default='near_sym_b', help='DTCWT biorthogonal filter (default: near_sym_b)') + parser.add_argument('--qshift', default='qshift_b', help='DTCWT quarter-shift filter (default: qshift_b)') args = parser.parse_args() pipeline_start = time.time() @@ -352,7 +368,11 @@ def main(): print(f"frame_count: {frame_count}, frame_width: {frame_width}, frame_height: {frame_height}, fps: {fps}") - nlevels = 3 + nlevels = args.nlevels + if nlevels < 1: + print("Error: --nlevels must be >= 1") + cap.release() + sys.exit(1) min_dim = 2 ** nlevels if roi is not None: @@ -377,9 +397,29 @@ def main(): ref_orient = 0 if args.gpu: - sound_data = extract_audio_gpu(cap, frame_count, nlevels, n_orient, ref_index, ref_orient, ref_level, fps, freq_low, freq_high, roi, args.batch_size) + import torch + proc_h = roi[3] if roi else frame_height + proc_w = roi[2] if roi else frame_width + required = estimate_vram(args.batch_size, proc_h, proc_w, nlevels) + free, total = torch.cuda.mem_get_info(0) + required_gb = required / (1024 ** 3) + free_gb = free / (1024 ** 3) + total_gb = total / (1024 ** 3) + print(f" Estimated VRAM needed: {required_gb:.1f} GB") + print(f" GPU VRAM available: {free_gb:.1f} GB / {total_gb:.1f} GB") + if required > free * 0.7: + print( + f"\nWarning: estimated VRAM ({required_gb:.1f} GB) exceeds 70% of " + f"available ({free_gb:.1f} GB).\n" + f" Suggestions:\n" + f" - Reduce --batch-size (current: {args.batch_size})\n" + f" - Use --roi to crop to a smaller region\n" + f" - Remove --gpu to use CPU mode", + file=sys.stderr + ) + sound_data = extract_audio_gpu(cap, frame_count, nlevels, n_orient, ref_index, ref_orient, ref_level, fps, freq_low, freq_high, roi, args.batch_size, args.biort, args.qshift) else: - sound_data = extract_audio(cap, frame_count, nlevels, n_orient, ref_index, ref_orient, ref_level, fps, freq_low, freq_high, roi) + sound_data = extract_audio(cap, frame_count, nlevels, n_orient, ref_index, ref_orient, ref_level, fps, freq_low, freq_high, roi, args.biort, args.qshift) save_wav(sound_data, output_name, int(fps)) print(f"Total time: {format_duration(time.time() - pipeline_start)}") From ab7460164956cf0f4a0600e8f34b7376710271fd Mon Sep 17 00:00:00 2001 From: Joel Jose Date: Sat, 21 Mar 2026 16:46:32 +0530 Subject: [PATCH 3/6] Upgrade GPU Dockerfile to PyTorch 2.1.2/CUDA 12.1 and add version tagging Upgrade GPU base image from pytorch:1.12.1-cuda11.3 to pytorch:2.1.2-cuda12.1-cudnn8-runtime. Pin numpy<2 in requirements-gpu.txt for pytorch_wavelets compatibility. Pin upper bounds on all dependencies. Update both Dockerfiles to include tests/ and requirements-dev.txt. Add VERSION build arg and version label. Update docker-build scripts to read VERSION file and tag images with version + latest. Add requirements-dev.txt (pytest, ruff) and tests/ stub for Dockerfile compatibility. Fixes #5 --- Dockerfile | 10 ++++++++-- Dockerfile.gpu | 15 ++++++++++++--- docker-build-gpu.sh | 9 ++++++--- docker-build.sh | 8 ++++++-- requirements-dev.txt | 2 ++ requirements-gpu.txt | 6 +++--- requirements.txt | 8 ++++---- tests/__init__.py | 0 8 files changed, 41 insertions(+), 17 deletions(-) create mode 100644 requirements-dev.txt create mode 100644 tests/__init__.py diff --git a/Dockerfile b/Dockerfile index 1468159..f9ad265 100644 --- a/Dockerfile +++ b/Dockerfile @@ -13,10 +13,16 @@ RUN groupadd -g ${GID} ${UNAME} && \ WORKDIR /app -COPY requirements.txt . -RUN pip install --no-cache-dir -r requirements.txt +COPY requirements.txt requirements-dev.txt ./ +RUN pip install --no-cache-dir -r requirements.txt -r requirements-dev.txt COPY visualmic.py . +COPY tests/ tests/ + +RUN chown -R ${UID}:${GID} /app + +ARG VERSION +LABEL version=${VERSION} USER ${UNAME} diff --git a/Dockerfile.gpu b/Dockerfile.gpu index 956bd83..9567eef 100644 --- a/Dockerfile.gpu +++ b/Dockerfile.gpu @@ -1,4 +1,7 @@ -FROM pytorch/pytorch:1.12.1-cuda11.3-cudnn8-runtime +# Pin PyTorch 2.1.2 because pytorch_wavelets is unmaintained (last commit 2023) +# and uses old-style autograd.Function. Tested working with 2.1.2. +# numpy<2 required because pytorch_wavelets uses removed NumPy 2.0 APIs. +FROM pytorch/pytorch:2.1.2-cuda12.1-cudnn8-runtime RUN apt-get update && \ apt-get install -y --no-install-recommends libgl1 libglib2.0-0 git && \ @@ -13,11 +16,17 @@ RUN groupadd -g ${GID} ${UNAME} && \ WORKDIR /app -COPY requirements-gpu.txt . -RUN pip install --no-cache-dir -r requirements-gpu.txt && \ +COPY requirements-gpu.txt requirements-dev.txt ./ +RUN pip install --no-cache-dir -r requirements-gpu.txt -r requirements-dev.txt && \ pip install --no-cache-dir git+https://github.com/fbcotter/pytorch_wavelets.git COPY visualmic.py . +COPY tests/ tests/ + +RUN chown -R ${UID}:${GID} /app + +ARG VERSION +LABEL version=${VERSION} USER ${UNAME} diff --git a/docker-build-gpu.sh b/docker-build-gpu.sh index ab918d0..8c22d1c 100755 --- a/docker-build-gpu.sh +++ b/docker-build-gpu.sh @@ -1,12 +1,15 @@ #!/bin/bash set -e +VERSION=$(cat VERSION) + docker build \ - --network=host \ --build-arg UID="$(id -u)" \ --build-arg GID="$(id -g)" \ --build-arg UNAME="$(whoami)" \ + --build-arg VERSION="${VERSION}" \ -f Dockerfile.gpu \ - -t visual-mic-gpu . + -t visual-mic-gpu:${VERSION} \ + -t visual-mic-gpu:latest . -echo "Built visual-mic-gpu image as user: $(whoami) (uid=$(id -u), gid=$(id -g))" +echo "Built visual-mic-gpu:${VERSION} (also tagged :latest)" diff --git a/docker-build.sh b/docker-build.sh index b032aa8..7faebb6 100755 --- a/docker-build.sh +++ b/docker-build.sh @@ -1,10 +1,14 @@ #!/bin/bash set -e +VERSION=$(cat VERSION) + docker build \ --build-arg UID="$(id -u)" \ --build-arg GID="$(id -g)" \ --build-arg UNAME="$(whoami)" \ - -t visual-mic . + --build-arg VERSION="${VERSION}" \ + -t visual-mic:${VERSION} \ + -t visual-mic:latest . -echo "Built visual-mic image as user: $(whoami) (uid=$(id -u), gid=$(id -g))" +echo "Built visual-mic:${VERSION} (also tagged :latest)" diff --git a/requirements-dev.txt b/requirements-dev.txt new file mode 100644 index 0000000..f643047 --- /dev/null +++ b/requirements-dev.txt @@ -0,0 +1,2 @@ +pytest>=7.0,<9 +ruff>=0.4.0,<1 diff --git a/requirements-gpu.txt b/requirements-gpu.txt index f15397b..8711e83 100644 --- a/requirements-gpu.txt +++ b/requirements-gpu.txt @@ -1,4 +1,4 @@ -scipy>=1.7.1 -numpy>=1.20.3 -opencv-python-headless>=4.7.0 +scipy>=1.7.1,<2 +numpy>=1.20.3,<2 +opencv-python-headless>=4.7.0,<5 PyWavelets>=1.1.0 diff --git a/requirements.txt b/requirements.txt index 201b62a..0da6bd3 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,4 +1,4 @@ -scipy>=1.7.1 -numpy>=1.20.3 -dtcwt>=0.12.0 -opencv-python>=4.7.0 +scipy>=1.7.1,<2 +numpy>=1.20.3,<3 +dtcwt>=0.12.0,<1 +opencv-python>=4.7.0,<5 diff --git a/tests/__init__.py b/tests/__init__.py new file mode 100644 index 0000000..e69de29 From 28e746adbb387b21b440e47d959b8aee66750f48 Mon Sep 17 00:00:00 2001 From: Joel Jose Date: Sat, 21 Mar 2026 16:52:07 +0530 Subject: [PATCH 4/6] Add testing infrastructure with CPU and GPU test suites Add tests/test_visualmic.py (28 CPU tests) with tiered coverage: - Strict: format_duration, find_best_shift, save_wav - Moderate: postprocess_phase_signals, Butterworth filter, estimate_vram - Smoke: extract_audio on synthetic 256x256 video, CLI validation Add tests/test_visualmic_gpu.py (7 GPU tests): - DTCWTForward shapes/finiteness, custom filters - extract_audio_gpu smoke test on synthetic video - All skip automatically without CUDA Add test.sh Docker-based test runner supporting cpu/gpu/--build modes. Move nlevels validation before file check for early error reporting. Fixes #6 --- test.sh | 44 ++++++ tests/test_visualmic.py | 289 ++++++++++++++++++++++++++++++++++++ tests/test_visualmic_gpu.py | 114 ++++++++++++++ visualmic.py | 13 +- 4 files changed, 453 insertions(+), 7 deletions(-) create mode 100755 test.sh create mode 100644 tests/test_visualmic.py create mode 100644 tests/test_visualmic_gpu.py diff --git a/test.sh b/test.sh new file mode 100755 index 0000000..d562b10 --- /dev/null +++ b/test.sh @@ -0,0 +1,44 @@ +#!/bin/bash +set -e + +MODE="${1:-cpu}" +BUILD_FLAG="${2:-}" + +if [[ "$MODE" == "gpu" ]]; then + IMAGE="visual-mic-gpu-dev" + DOCKERFILE="-f Dockerfile.gpu" + RUN_FLAGS="--gpus device=0" +elif [[ "$MODE" == "--build" ]]; then + # Handle ./test.sh --build (no mode, just build flag) + MODE="cpu" + BUILD_FLAG="--build" + IMAGE="visual-mic-dev" + DOCKERFILE="" + RUN_FLAGS="" +else + IMAGE="visual-mic-dev" + DOCKERFILE="" + RUN_FLAGS="" +fi + +# Build image if it doesn't exist or --build flag passed +if [[ "$BUILD_FLAG" == "--build" ]] || ! docker image inspect ${IMAGE} &>/dev/null; then + echo "Building test image (${MODE})..." + docker build \ + --build-arg UID="$(id -u)" \ + --build-arg GID="$(id -g)" \ + --build-arg UNAME="$(whoami)" \ + ${DOCKERFILE} \ + -t ${IMAGE} . + echo "" +fi + +echo "=== Lint ===" +docker run --rm --entrypoint "" ${IMAGE} ruff check . + +echo "" +echo "=== Tests ===" +docker run --rm ${RUN_FLAGS} --entrypoint "" ${IMAGE} python -m pytest tests/ -v + +echo "" +echo "All checks passed." diff --git a/tests/test_visualmic.py b/tests/test_visualmic.py new file mode 100644 index 0000000..fbf0824 --- /dev/null +++ b/tests/test_visualmic.py @@ -0,0 +1,289 @@ +"""Unit tests for visualmic.py — CPU path.""" + +import os +import subprocess +import sys + +import cv2 +import numpy as np +import pytest + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) +import visualmic + + +# --- Tier 1: Strict (exact equality) --- + + +class TestFormatDuration: + def test_zero(self): + assert visualmic.format_duration(0) == "0s" + + def test_seconds(self): + assert visualmic.format_duration(59) == "59s" + + def test_one_minute(self): + assert visualmic.format_duration(60) == "1m 0s" + + def test_minutes_seconds(self): + assert visualmic.format_duration(125) == "2m 5s" + + def test_hours(self): + assert visualmic.format_duration(3661) == "1h 1m 1s" + + def test_fractional_truncates(self): + assert visualmic.format_duration(59.9) == "59s" + + +class TestFindBestShift: + def test_known_shift(self): + # b is shifted right by 10 relative to a → find_best_shift returns -10 + # (the shift needed to align b back to a) + a = np.zeros(200) + a[50:60] = 1.0 + b = np.zeros(200) + b[60:70] = 1.0 + result = visualmic.find_best_shift(a, b) + assert result == -10 + + def test_zero_shift(self): + a = np.zeros(100) + a[30:40] = 1.0 + result = visualmic.find_best_shift(a, a) + assert result == 0 + + def test_negative_shift(self): + # b is shifted left by 5 relative to a → find_best_shift returns 5 + a = np.zeros(200) + a[60:70] = 1.0 + b = np.zeros(200) + b[55:65] = 1.0 + result = visualmic.find_best_shift(a, b) + assert result == 5 + + +class TestSaveWav: + def test_creates_file(self, tmp_path): + samples = np.array([0.0, 0.5, -0.5, 1.0, -1.0]) + path = str(tmp_path / "test.wav") + visualmic.save_wav(samples, path, 44100) + assert os.path.isfile(path) + + def test_sample_rate(self, tmp_path): + from scipy.io.wavfile import read as read_wav + samples = np.array([0.0, 0.5, -0.5]) + path = str(tmp_path / "test.wav") + visualmic.save_wav(samples, path, 2200) + sr, _ = read_wav(path) + assert sr == 2200 + + def test_int16_range(self, tmp_path): + from scipy.io.wavfile import read as read_wav + samples = np.array([1.0, -1.0, 0.0]) + path = str(tmp_path / "test.wav") + visualmic.save_wav(samples, path, 44100) + _, data = read_wav(path) + assert data.dtype == np.int16 + assert np.max(data) == 32767 + assert np.min(data) == -32767 + + +# --- Tier 2: Moderate tolerance --- + + +class TestPostprocessPhaseSignals: + def test_constant_input_silent(self): + """Constant phase signals (no motion) should produce silent output.""" + nlevels, n_orient, frame_count = 3, 6, 50 + phase_signals = np.ones((frame_count, nlevels, n_orient)) + result = visualmic.postprocess_phase_signals( + phase_signals, frame_count, nlevels, n_orient, + ref_level=0, ref_orient=0, fps=100 + ) + # Constant input → all sub-bands identical → after shift+sum, still constant + # Normalization maps constant to zero + assert result.shape == (frame_count,) + assert np.all(np.isfinite(result)) + + def test_normalization_range(self): + """Output should be in [-1, 1].""" + nlevels, n_orient, frame_count = 3, 6, 100 + rng = np.random.RandomState(42) + phase_signals = rng.randn(frame_count, nlevels, n_orient) + result = visualmic.postprocess_phase_signals( + phase_signals, frame_count, nlevels, n_orient, + ref_level=0, ref_orient=0, fps=100 + ) + assert np.min(result) >= -1.0 - 1e-10 + assert np.max(result) <= 1.0 + 1e-10 + + def test_output_shape(self): + nlevels, n_orient, frame_count = 2, 6, 30 + rng = np.random.RandomState(42) + phase_signals = rng.randn(frame_count, nlevels, n_orient) + result = visualmic.postprocess_phase_signals( + phase_signals, frame_count, nlevels, n_orient, + ref_level=0, ref_orient=0, fps=100 + ) + assert result.shape == (frame_count,) + + +class TestButterworthFilter: + def test_passband_preserved(self): + """Signal within passband should be preserved.""" + nlevels, n_orient, frame_count = 1, 1, 200 + fps = 1000 + # 100 Hz signal, passband 50-200 Hz + t = np.arange(frame_count) / fps + phase_signals = np.sin(2 * np.pi * 100 * t).reshape(-1, 1, 1) + result = visualmic.postprocess_phase_signals( + phase_signals.copy(), frame_count, nlevels, n_orient, + ref_level=0, ref_orient=0, fps=fps, freq_low=50, freq_high=200 + ) + assert np.all(np.isfinite(result)) + # Should have non-zero energy (signal passed through) + assert np.std(result) > 0.01 + + def test_stopband_attenuated(self): + """Signal outside passband should be attenuated.""" + nlevels, n_orient, frame_count = 1, 1, 200 + fps = 1000 + # 400 Hz signal, passband 50-200 Hz + t = np.arange(frame_count) / fps + phase_signals = np.sin(2 * np.pi * 400 * t).reshape(-1, 1, 1) + result = visualmic.postprocess_phase_signals( + phase_signals.copy(), frame_count, nlevels, n_orient, + ref_level=0, ref_orient=0, fps=fps, freq_low=50, freq_high=200 + ) + assert np.all(np.isfinite(result)) + # Should have lower energy than unfiltered version + unfiltered = visualmic.postprocess_phase_signals( + phase_signals.copy(), frame_count, nlevels, n_orient, + ref_level=0, ref_orient=0, fps=fps + ) + assert np.std(result) < np.std(unfiltered) + + +class TestEstimateVram: + def test_basic_arithmetic(self): + result = visualmic.estimate_vram(16, 256, 256, 3) + # 16 * 256 * 256 * 4 * 15 + 300MB + expected = 16 * 256 * 256 * 4 * 15 + 300 * 1024 * 1024 + assert result == expected + + def test_scales_with_batch_size(self): + small = visualmic.estimate_vram(8, 256, 256, 3) + large = visualmic.estimate_vram(32, 256, 256, 3) + assert large > small + + def test_scales_with_resolution(self): + small = visualmic.estimate_vram(16, 256, 256, 3) + large = visualmic.estimate_vram(16, 512, 512, 3) + assert large > small + + +# --- Tier 3: Smoke tests --- + + +def _create_synthetic_video(path, num_frames=32, height=256, width=256): + """Create a synthetic video with random noise for testing.""" + fourcc = cv2.VideoWriter_fourcc(*'MJPG') + out = cv2.VideoWriter(path, fourcc, 30.0, (width, height)) + rng = np.random.RandomState(42) + for _ in range(num_frames): + frame = rng.randint(0, 256, (height, width, 3), dtype=np.uint8) + out.write(frame) + out.release() + + +try: + import dtcwt # noqa: F401 + HAS_DTCWT = True +except ImportError: + HAS_DTCWT = False + + +@pytest.mark.skipif(not HAS_DTCWT, reason="dtcwt not available") +class TestExtractAudio: + def test_smoke_synthetic_video(self, tmp_path): + """Full pipeline on synthetic video: verify shape and finiteness.""" + video_path = str(tmp_path / "test.avi") + _create_synthetic_video(video_path, num_frames=32, height=256, width=256) + + cap = cv2.VideoCapture(video_path) + result = visualmic.extract_audio( + cap, 32, nlevels=3, n_orient=6, + ref_index=0, ref_orient=0, ref_level=0, + fps=30 + ) + assert result.shape == (32,) + assert np.all(np.isfinite(result)) + assert np.min(result) >= -1.0 - 1e-10 + assert np.max(result) <= 1.0 + 1e-10 + + def test_with_biort_qshift(self, tmp_path): + """Verify custom filter selection works.""" + video_path = str(tmp_path / "test.avi") + _create_synthetic_video(video_path, num_frames=16, height=256, width=256) + + cap = cv2.VideoCapture(video_path) + result = visualmic.extract_audio( + cap, 16, nlevels=2, n_orient=6, + ref_index=0, ref_orient=0, ref_level=0, + fps=30, biort='near_sym_a', qshift='qshift_a' + ) + assert result.shape == (16,) + assert np.all(np.isfinite(result)) + + +class TestInputValidation: + def test_missing_input(self): + result = subprocess.run( + [sys.executable, 'visualmic.py'], + capture_output=True, text=True + ) + assert result.returncode != 0 + + @pytest.mark.skipif(not HAS_DTCWT, reason="dtcwt not available") + def test_nonexistent_file(self): + result = subprocess.run( + [sys.executable, 'visualmic.py', '-i', 'nonexistent.avi'], + capture_output=True, text=True + ) + assert result.returncode != 0 + assert 'not found' in result.stderr + result.stdout + + def test_bad_frequencies(self): + result = subprocess.run( + [sys.executable, 'visualmic.py', '-i', 'dummy.avi', + '-fl', '200', '-fh', '100'], + capture_output=True, text=True + ) + assert result.returncode != 0 + assert 'freq-low' in result.stderr + result.stdout + + def test_bad_roi_format(self): + result = subprocess.run( + [sys.executable, 'visualmic.py', '-i', 'dummy.avi', + '--roi', '100,50,200'], + capture_output=True, text=True + ) + assert result.returncode != 0 + assert 'four integers' in result.stderr + result.stdout + + def test_invalid_nlevels(self): + result = subprocess.run( + [sys.executable, 'visualmic.py', '-i', 'dummy.avi', + '--nlevels', '0'], + capture_output=True, text=True + ) + assert result.returncode != 0 + assert 'nlevels' in result.stderr + result.stdout + + def test_version_flag(self): + result = subprocess.run( + [sys.executable, 'visualmic.py', '--version'], + capture_output=True, text=True + ) + assert result.returncode == 0 + assert '2.0.0' in result.stdout diff --git a/tests/test_visualmic_gpu.py b/tests/test_visualmic_gpu.py new file mode 100644 index 0000000..5d0a1d0 --- /dev/null +++ b/tests/test_visualmic_gpu.py @@ -0,0 +1,114 @@ +"""Unit tests for visualmic.py — GPU path (CUDA only).""" + +import os +import sys + +import cv2 +import numpy as np +import pytest + +try: + import torch + HAS_CUDA = torch.cuda.is_available() +except ImportError: + HAS_CUDA = False + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) +import visualmic + + +pytestmark = pytest.mark.skipif(not HAS_CUDA, reason="CUDA not available") + + +def _create_synthetic_video(path, num_frames=32, height=256, width=256): + """Create a synthetic video with random noise for testing.""" + fourcc = cv2.VideoWriter_fourcc(*'MJPG') + out = cv2.VideoWriter(path, fourcc, 30.0, (width, height)) + rng = np.random.RandomState(42) + for _ in range(num_frames): + frame = rng.randint(0, 256, (height, width, 3), dtype=np.uint8) + out.write(frame) + out.release() + + +class TestGpuForwardPass: + def test_dtcwt_forward_shapes(self): + """Verify DTCWTForward produces expected shapes.""" + from pytorch_wavelets import DTCWTForward + + xfm = DTCWTForward(J=3, biort='near_sym_b', qshift='qshift_b').cuda() + batch = torch.randn(4, 1, 256, 256, device='cuda') + Yl, Yh = xfm(batch) + + assert Yl.shape[0] == 4 + assert len(Yh) == 3 + for level in range(3): + assert Yh[level].shape[0] == 4 + assert Yh[level].shape[2] == 6 # 6 orientations + assert Yh[level].shape[-1] == 2 # real/imag + + def test_dtcwt_forward_finite(self): + """Verify all outputs are finite.""" + from pytorch_wavelets import DTCWTForward + + xfm = DTCWTForward(J=3, biort='near_sym_b', qshift='qshift_b').cuda() + batch = torch.randn(2, 1, 256, 256, device='cuda') + Yl, Yh = xfm(batch) + + assert torch.isfinite(Yl).all() + for level in range(3): + assert torch.isfinite(Yh[level]).all() + + def test_custom_filters(self): + """Verify custom biort/qshift filters work.""" + from pytorch_wavelets import DTCWTForward + + xfm = DTCWTForward(J=2, biort='near_sym_a', qshift='qshift_a').cuda() + batch = torch.randn(2, 1, 256, 256, device='cuda') + Yl, Yh = xfm(batch) + assert torch.isfinite(Yl).all() + + +class TestExtractAudioGpu: + def test_smoke_synthetic_video(self, tmp_path): + """Full GPU pipeline on synthetic video: verify shape and finiteness.""" + video_path = str(tmp_path / "test.avi") + _create_synthetic_video(video_path, num_frames=32, height=256, width=256) + + cap = cv2.VideoCapture(video_path) + result = visualmic.extract_audio_gpu( + cap, 32, nlevels=3, n_orient=6, + ref_index=0, ref_orient=0, ref_level=0, + fps=30, batch_size=16 + ) + assert result.shape == (32,) + assert np.all(np.isfinite(result)) + assert np.min(result) >= -1.0 - 1e-10 + assert np.max(result) <= 1.0 + 1e-10 + + def test_with_custom_filters(self, tmp_path): + """Verify GPU path works with custom filter selection.""" + video_path = str(tmp_path / "test.avi") + _create_synthetic_video(video_path, num_frames=16, height=256, width=256) + + cap = cv2.VideoCapture(video_path) + result = visualmic.extract_audio_gpu( + cap, 16, nlevels=2, n_orient=6, + ref_index=0, ref_orient=0, ref_level=0, + fps=30, batch_size=8, + biort='near_sym_a', qshift='qshift_a' + ) + assert result.shape == (16,) + assert np.all(np.isfinite(result)) + + +class TestEstimateVram: + def test_arithmetic(self): + result = visualmic.estimate_vram(16, 256, 256, 3) + expected = 16 * 256 * 256 * 4 * 15 + 300 * 1024 * 1024 + assert result == expected + + def test_scales_with_batch_size(self): + small = visualmic.estimate_vram(8, 256, 256, 3) + large = visualmic.estimate_vram(32, 256, 256, 3) + assert large > small diff --git a/visualmic.py b/visualmic.py index 66f2778..c316639 100644 --- a/visualmic.py +++ b/visualmic.py @@ -303,6 +303,10 @@ def main(): if freq_low is not None and freq_high is not None and freq_low >= freq_high: print(f"Error: freq-low ({freq_low} Hz) must be less than freq-high ({freq_high} Hz)") sys.exit(1) + nlevels = args.nlevels + if nlevels < 1: + print("Error: --nlevels must be >= 1") + sys.exit(1) roi = None if args.roi is not None: try: @@ -324,13 +328,13 @@ def main(): print("Error: --gpu requires PyTorch (pip install torch)") sys.exit(1) try: - import pytorch_wavelets + import pytorch_wavelets # noqa: F401 except ImportError: print("Error: --gpu requires pytorch_wavelets (pip install git+https://github.com/fbcotter/pytorch_wavelets.git)") sys.exit(1) else: try: - import dtcwt + import dtcwt # noqa: F401 except ImportError: print("Error: CPU mode requires dtcwt (pip install dtcwt)") sys.exit(1) @@ -368,11 +372,6 @@ def main(): print(f"frame_count: {frame_count}, frame_width: {frame_width}, frame_height: {frame_height}, fps: {fps}") - nlevels = args.nlevels - if nlevels < 1: - print("Error: --nlevels must be >= 1") - cap.release() - sys.exit(1) min_dim = 2 ** nlevels if roi is not None: From 170b7f8cd8500a2d55e72d91a15b5ce21d3e7e9e Mon Sep 17 00:00:00 2001 From: Joel Jose Date: Sat, 21 Mar 2026 16:52:31 +0530 Subject: [PATCH 5/6] Add CI/CD pipeline with CPU and GPU jobs Add .github/workflows/ci.yml with two jobs matching EVM/DTCWT repos: - CPU: build Docker, lint (ruff), verify import/help/version - GPU: build GPU Docker, lint (ruff), verify import/help/version Triggers on push to main and pull requests to main. Fixes #7 --- .github/workflows/ci.yml | 58 ++++++++++++++++++++++++++++++++++++++++ 1 file changed, 58 insertions(+) create mode 100644 .github/workflows/ci.yml diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..38f2de0 --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,58 @@ +name: CI + +on: + push: + branches: [main] + pull_request: + branches: [main] + +jobs: + test: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v6 + + - name: Build Docker image + run: | + docker build \ + --build-arg UID=$(id -u) \ + --build-arg GID=$(id -g) \ + --build-arg UNAME=$(whoami) \ + -t visualmic-ci . + + - name: Lint + run: docker run --rm --entrypoint "" visualmic-ci ruff check . + + - name: Verify import + run: docker run --rm --entrypoint "" visualmic-ci python -c "import visualmic" + + - name: Verify --help + run: docker run --rm --entrypoint "" visualmic-ci python visualmic.py --help + + - name: Verify --version + run: docker run --rm --entrypoint "" visualmic-ci python visualmic.py --version + + test-gpu: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v6 + + - name: Build GPU Docker image + run: | + docker build -f Dockerfile.gpu \ + --build-arg UID=$(id -u) \ + --build-arg GID=$(id -g) \ + --build-arg UNAME=$(whoami) \ + -t visualmic-gpu-ci . + + - name: Lint + run: docker run --rm --entrypoint "" visualmic-gpu-ci ruff check . + + - name: Verify import + run: docker run --rm --entrypoint "" visualmic-gpu-ci python -c "import visualmic" + + - name: Verify --help + run: docker run --rm --entrypoint "" visualmic-gpu-ci python visualmic.py --help + + - name: Verify --version + run: docker run --rm --entrypoint "" visualmic-gpu-ci python visualmic.py --version From cc0acc6627aec9ad90dc63fc1b3066f184cddf36 Mon Sep 17 00:00:00 2001 From: Joel Jose Date: Sat, 21 Mar 2026 16:57:18 +0530 Subject: [PATCH 6/6] Add docs and clean up README for v2.0.0 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add CHANGELOG.md (Keep a Changelog, starting at v2.0.0), CONTRIBUTING.md, and docs/design/visualmic-hardening.md ADR. README overhaul: - Remove Part 4 (Denoising) — use audio_denoising repo instead - Restructure Setup section (remove YouTube link, add GPU Docker) - Add CI badge - Add Development section (tests, versioning, project structure) - Update CLI flags table with --nlevels, --biort, --qshift, --version - Update Future Work (remove GPU as done, remove denoising ref) - Update parameters table with filter defaults Fixes #8 --- CHANGELOG.md | 31 ++++ CONTRIBUTING.md | 21 +++ README.md | 270 ++++++++++++++++++----------- docs/design/visualmic-hardening.md | 268 ++++++++++++++++++++++++++++ 4 files changed, 492 insertions(+), 98 deletions(-) create mode 100644 CHANGELOG.md create mode 100644 CONTRIBUTING.md create mode 100644 docs/design/visualmic-hardening.md diff --git a/CHANGELOG.md b/CHANGELOG.md new file mode 100644 index 0000000..55de2d7 --- /dev/null +++ b/CHANGELOG.md @@ -0,0 +1,31 @@ +# Changelog + +All notable changes to this project will be documented in this file. + +The format is based on [Keep a Changelog](https://keepachangelog.com/), +and this project adheres to [Semantic Versioning](https://semver.org/). + +## [Unreleased] + +## [2.0.0] - 2026-03-21 + +### Removed +- Denoising (`--denoise`, `--denoise-input`). Use the standalone [`audio_denoising`](https://github.com/joeljose/audio_denoising) tool instead. + +### Added +- `--nlevels` CLI flag for configurable DTCWT decomposition levels (default: 3) +- `--biort` and `--qshift` CLI flags for wavelet filter selection (default: `near_sym_b`/`qshift_b`) +- `--version` flag +- Pre-flight GPU VRAM estimation and warning +- Unit tests for CPU and GPU paths (`tests/test_visualmic.py`, `tests/test_visualmic_gpu.py`) +- Docker-based test runner (`test.sh`) supporting cpu/gpu modes +- CI/CD pipeline (`.github/workflows/ci.yml`) +- `CHANGELOG.md`, `CONTRIBUTING.md`, design docs +- `VERSION` file as single source of truth for versioning +- Docker images tagged with version numbers + +### Changed +- Default wavelet filters changed to `near_sym_b`/`qshift_b` for both CPU and GPU paths (matching DTCWT Motion Mag v2.0.0). Restore old behavior with `--biort near_sym_a --qshift qshift_a`. +- GPU Dockerfile upgraded from PyTorch 1.12.1/CUDA 11.3 to PyTorch 2.1.2/CUDA 12.1 +- `-i`/`--input` is now always required (previously optional when using `--denoise-input`) +- Pinned upper bounds on all dependencies diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md new file mode 100644 index 0000000..c9b7efc --- /dev/null +++ b/CONTRIBUTING.md @@ -0,0 +1,21 @@ +# Contributing + +Thanks for your interest in contributing! + +## How to contribute + +1. **Open an issue first** — describe the bug or feature before writing code. No silent PRs. +2. **Fork and branch** — create a feature branch from `main`. +3. **Keep PRs small** — one logical change per PR. +4. **Follow PEP 8** — enforced by `ruff` in CI. +5. **Test before opening a PR** — run `./test.sh` (and `./test.sh gpu` if touching GPU code). + +## Running tests + +All tests run inside Docker — no local Python dependencies needed: + +```bash +./test.sh # CPU: lint + unit tests +./test.sh gpu # GPU: lint + unit tests (requires nvidia-container-toolkit) +./test.sh --build # Force rebuild image before testing +``` diff --git a/README.md b/README.md index b9430a3..55ef489 100644 --- a/README.md +++ b/README.md @@ -1,5 +1,7 @@ # Visual-Mic +[![CI](https://github.com/joeljose/Visual-Mic/actions/workflows/ci.yml/badge.svg)](https://github.com/joeljose/Visual-Mic/actions/workflows/ci.yml) + A Python implementation of the [Visual Microphone](https://people.csail.mit.edu/mrub/VisualMic/) algorithm, which recovers sound from high-speed video by analyzing sub-pixel surface vibrations. When sound hits an object, it causes tiny vibrations on the surface—far too small to see with the naked eye, but detectable in the phase of complex wavelet coefficients. This tool extracts those vibrations and reconstructs an audible signal, effectively turning everyday objects into microphones. The original work by [Davis et al. (MIT CSAIL, SIGGRAPH 2014)](https://people.csail.mit.edu/mrub/papers/VisualMic_SIGGRAPH2014.pdf) used Complex Steerable Pyramids for the video decomposition. This project uses **2D Dual-Tree Complex Wavelet Transform (DTCWT)** instead, which is ~5x more computationally efficient while still providing reliable phase information for motion estimation. We test against the same high-speed videos provided by MIT CSAIL. @@ -12,7 +14,13 @@ The sample videos can be downloaded from [here](http://data.csail.mit.edu/vidmag ## Table of Contents -- [Setting Up](#setting-up-visual-mic) +- [Setup](#setup) + - [A. Local Setup](#a-local-setup) + - [B. Docker (CPU)](#b-docker-cpu) + - [C. Docker (GPU)](#c-docker-gpu) +- [Usage](#usage) + - [CLI Tool](#cli-tool) + - [Tips](#tips) - [Part 1: The Original Work (Davis et al., SIGGRAPH 2014)](#part-1-the-original-work-davis-et-al-siggraph-2014) - [1.1 The Physical Phenomenon](#11-the-physical-phenomenon) - [1.2 Why Not Just Track Pixels?](#12-why-not-just-track-pixels) @@ -28,95 +36,117 @@ The sample videos can be downloaded from [here](http://data.csail.mit.edu/vidmag - [2.4 Parameters](#24-parameters-used) - [2.5 What Each Scale Captures](#25-what-each-scale-captures) - [Part 3: Literature Survey](#part-3-literature-survey) -- [Part 4: Denoising](#part-4-denoising) - [Future Work](#future-work) +- [Development](#development) + - [Running Tests](#running-tests) + - [Versioning](#versioning) + - [Project Structure](#project-structure) - [References](#references) --- -## Setting up visual mic +## Setup + +### A. Local Setup + +```bash +git clone https://github.com/joeljose/Visual-Mic.git +cd Visual-Mic +pip install -r requirements.txt +python visualmic.py -i testvid.avi -o recovered_audio.wav +``` + +**Requirements:** Python 3.8+ + +### B. Docker (CPU) + +```bash +# Build +./docker-build.sh + +# Run +docker run --rm -v /path/to/videos:/data \ + visual-mic:latest \ + -i /data/testvid.avi -o /data/sound.wav +``` + +### C. Docker (GPU) + +Requires [nvidia-container-toolkit](https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/install-guide.html). + +```bash +# Build +./docker-build-gpu.sh -### A. Setting up Python3(skip if already setup) +# Run +docker run --rm --gpus all -v /path/to/videos:/data \ + visual-mic-gpu:latest \ + --gpu -i /data/Chips1-2200Hz-Mary_Had-input.avi \ + -o /data/sound.wav --fps 2200 --batch-size 32 +``` -You can follow this link from [Youtube](https://www.youtube.com/watch?v=YYXdXT2l-Gg). This has a very concise explanation on how to setup python. +The `--batch-size` flag controls how many frames are processed per GPU batch (default: 16). Larger batches are faster but use more GPU memory. At 704x704, each frame uses ~10 MB of GPU memory, so `--batch-size 32` needs ~660 MB including overhead. -### B. Rest of the setup +**Note:** GPU mode produces very similar but not bit-identical output compared to CPU mode, due to float32 vs float64 precision differences and different DTCWT implementations (`pytorch_wavelets` vs `dtcwt`). -1. Clone the repo - ```sh - git clone https://github.com/joeljose/Visual-Mic.git - ``` -2. Navigate to "Visual-Mic" repo. -3. pip install all the python modules from requirements.txt(you should be in the "Visual-Mic" repository when you execute this command.) - ```sh - pip install -r requirements.txt - ``` -4. Run visualmic.py: - ```sh - python visualmic.py -i - python visualmic.py -i testvid.avi -o recovered_audio.wav - python visualmic.py -i testvid.avi -fl 80 -fh 1000 - python visualmic.py -i testvid.avi --roi 100,50,200,150 - python visualmic.py -i Chips1-2200Hz-Mary_Had-input.avi --fps 2200 - ``` - | Argument | Required | Description | - |----------|----------|-------------| - | `-i`, `--input` | Yes* | Path to input video file (*not required when using `--denoise-input`) | - | `-o`, `--output` | No | Output audio path (default: `sound.wav`) | - | `-fl`, `--freq-low` | No | Lower cutoff frequency in Hz for temporal bandpass filter | - | `-fh`, `--freq-high` | No | Upper cutoff frequency in Hz for temporal bandpass filter | - | `--fps` | No | Override the video frame rate (Hz) for audio output sample rate | - | `--roi` | No | Region of interest as `x,y,w,h` — crops each frame before processing | - | `--gpu` | No | Use GPU-accelerated DTCWT (requires CUDA and `pytorch_wavelets`) | - | `--batch-size` | No | Frames per GPU batch (default: 16, GPU mode only) | - | `--denoise` | No | Audio denoising: `spectral` (spectral subtraction) or `morphological` (spectrogram morphology) | - | `--denoise-input` | No | Denoise an existing WAV file instead of processing video | +--- - When `-fl` and/or `-fh` are specified, a Butterworth filter is applied to the phase signals before audio reconstruction, rejecting low-frequency drift and high-frequency noise to improve output quality. +## Usage - When `--fps` is specified, the given value is used as the audio sample rate instead of the frame rate reported by the video container. This is necessary for high-speed camera footage where the container frame rate does not reflect the actual capture rate. For example, the MIT CSAIL Chips1 video was captured at 2200 frames per second, but the AVI container reports ~30 fps. Without `--fps 2200`, the output audio would be sampled at 30 Hz and unplayable. +### CLI Tool - When `--roi` is specified, each frame is cropped to the given rectangle before the DTCWT decomposition. This reduces computation and can improve SNR by focusing on the vibrating object (e.g., the bag of chips) and excluding background regions. +```bash +# Basic usage +python visualmic.py -i testvid.avi -o recovered_audio.wav -### C. Running with Docker (alternative) +# With temporal bandpass filter +python visualmic.py -i testvid.avi -fl 80 -fh 1000 -No Python setup needed — just Docker. +# With ROI (focus on vibrating object) +python visualmic.py -i testvid.avi --roi 100,50,200,150 -1. Build the image (automatically picks up your username, UID, and GID): - ```sh - ./docker-build.sh - ``` - This builds a Docker image named **`visual-mic`** using `docker-build.sh`, which auto-detects your host username, UID, and GID so that output files are owned by your host user. +# Override frame rate for high-speed video +python visualmic.py -i Chips1-2200Hz-Mary_Had-input.avi --fps 2200 -2. Run (mount the directory containing your video): - ```sh - docker run --rm --name visual-mic-run -v /path/to/videos:/data visual-mic -i /data/testvid.avi -o /data/sound.wav - ``` - All the same arguments (`-fl`, `-fh`, `--fps`, `--roi`, etc.) work exactly as described above. +# GPU acceleration +python visualmic.py -i testvid.avi --gpu --batch-size 32 -### D. Running with GPU acceleration (Docker + CUDA) +# Custom wavelet filters +python visualmic.py -i testvid.avi --biort near_sym_a --qshift qshift_a +``` -For large videos, the DTCWT forward pass is the main bottleneck. GPU mode uses [`pytorch_wavelets`](https://github.com/fbcotter/pytorch_wavelets) with CUDA to run batched transforms on the GPU, providing a significant speedup. +| Flag | Default | Description | +|---|---|---| +| `-i / --input` | *(required)* | Input video path | +| `-o / --output` | `sound.wav` | Output audio path | +| `-fl / --freq-low` | — | Lower cutoff frequency (Hz) for temporal bandpass filter | +| `-fh / --freq-high` | — | Upper cutoff frequency (Hz) for temporal bandpass filter | +| `--fps` | — | Override video frame rate (Hz) for audio sample rate | +| `--roi` | — | Region of interest as `x,y,w,h` | +| `--gpu` | off | Use GPU-accelerated DTCWT (requires CUDA + pytorch_wavelets) | +| `--batch-size` | 16 | Frames per GPU batch (GPU mode only) | +| `--nlevels` | 3 | Number of DTCWT decomposition levels | +| `--biort` | `near_sym_b` | Biorthogonal wavelet filter for DTCWT level 1 | +| `--qshift` | `qshift_b` | Quarter-shift wavelet filter for DTCWT levels 2+ | +| `--version` | — | Show program version and exit | -**Requirements:** NVIDIA GPU with CUDA support, Docker with `--gpus` (nvidia-container-toolkit). +**Available wavelet filters:** +- `--biort`: `antonini`, `legall`, `near_sym_a`, `near_sym_b` +- `--qshift`: `qshift_06`, `qshift_a`, `qshift_b`, `qshift_c`, `qshift_d` -1. Build the GPU image: - ```sh - ./docker-build-gpu.sh - ``` +When `-fl` and/or `-fh` are specified, a Butterworth filter is applied to the phase signals before audio reconstruction, rejecting low-frequency drift and high-frequency noise. -2. Run with `--gpu`: - ```sh - docker run --rm --gpus device=0 -v /path/to/videos:/data \ - visual-mic-gpu --gpu -i /data/Chips1-2200Hz-Mary_Had-input.avi \ - -o /data/sound_gpu.wav --fps 2200 --batch-size 32 - ``` +When `--fps` is specified, the given value is used as the audio sample rate instead of the frame rate reported by the video container. This is necessary for high-speed camera footage where the container frame rate does not reflect the actual capture rate. - The `--batch-size` flag controls how many frames are processed per GPU batch (default: 16). Larger batches are faster but use more GPU memory. At 704x704, each frame uses ~10 MB of GPU memory, so `--batch-size 32` needs ~660 MB including overhead — well within the capacity of most GPUs. +When `--roi` is specified, each frame is cropped to the given rectangle before the DTCWT decomposition. This reduces computation and can improve SNR by focusing on the vibrating object. - If you run out of GPU memory, reduce `--batch-size` (e.g., `--batch-size 8` or `--batch-size 1`). +### Tips - **Note:** GPU mode produces very similar but not bit-identical output compared to CPU mode, due to float32 vs float64 precision differences. +- Start with default settings and adjust from there. +- Use `--roi` to focus on the vibrating object — improves SNR and reduces computation. +- For MIT CSAIL videos, always use `--fps 2200` (the container reports ~30 fps incorrectly). +- Use `--gpu` for large videos — the DTCWT forward pass is the main bottleneck. +- If GPU runs out of memory, reduce `--batch-size`. --- @@ -362,7 +392,7 @@ This is fewer orientations than a typical steerable pyramid (which might use 8+) Here's how `visualmic.py` implements the pipeline, with line references. -### Steps 1–3: Stream Video, ROI Crop, DTCWT, and Phase Extraction (lines 142–194) +### Steps 1–3: Stream Video, ROI Crop, DTCWT, and Phase Extraction Frames are streamed directly from the video file — each frame is read, transformed, and discarded immediately, so only one raw frame is in memory at a time. This enables processing of arbitrarily long videos without running out of memory. If an ROI is specified, each frame is cropped before the DTCWT decomposition, reducing computation and focusing on the vibrating object. @@ -396,10 +426,6 @@ def extract_audio(cap, frame_count, nlevels, n_orient, ref_index, ref_orient, re phase_signals = np.array(phase_signals) # shape: (frame_count, nlevels, n_orient) ``` -`transform.forward()` returns a `Pyramid` object: -- `pyramid.highpasses[level]` has shape $(H_{\text{level}}, W_{\text{level}}, 6)$ -- Each value is a **complex number** encoding amplitude and phase - Vectorized NumPy operations on entire 2D spatial slices: | Operation | Code | Corresponds to | @@ -413,7 +439,7 @@ The conjugate multiplication `coeffs * conj(ref)` computes the phase difference **Result:** `phase_signals[fc, level, angle]` $= \Phi(\text{level}, \text{angle}, fc)$ — one scalar per frame per sub-band. -### Step 3.5: Temporal Bandpass Filtering (lines 89–118, optional) +### Step 3.5: Temporal Bandpass Filtering (optional) When `-fl` and/or `-fh` are specified, a 4th-order Butterworth filter is applied to each of the 18 phase signals before cross-correlation: @@ -432,7 +458,7 @@ for i in range(nlevels): - Skipped if video has fewer than 13 frames (minimum required for `filtfilt`) - If only `-fl` is given, acts as highpass; if only `-fh`, acts as lowpass -### Step 4: Temporal Alignment via Cross-Correlation (lines 120–124) +### Step 4: Temporal Alignment via Cross-Correlation ```python ref_vector = phase_signals[:, ref_level, ref_orient].reshape(-1) @@ -441,7 +467,7 @@ for i in range(nlevels): shift_matrix[i, j] = find_best_shift(ref_vector, phase_signals[:, i, j].reshape(-1)) ``` -The `find_best_shift` function (lines 26–28) uses `scipy.signal.correlate` for $O(n \log n)$ cross-correlation: +The `find_best_shift` function uses `scipy.signal.correlate` for $O(n \log n)$ cross-correlation: ```python def find_best_shift(a, b): @@ -449,7 +475,7 @@ def find_best_shift(a, b): return np.argmax(correlation) - (len(b) - 1) ``` -### Step 5: Sum Across Sub-bands with Temporal Shifts (lines 126–129) +### Step 5: Sum Across Sub-bands with Temporal Shifts ```python sound_raw = np.zeros(frame_count) @@ -458,7 +484,7 @@ for i in range(nlevels): sound_raw += np.roll(phase_signals[:, i, j], int(shift_matrix[i, j])) ``` -### Step 6: Normalize to $[-1, 1]$ (lines 131–137) +### Step 6: Normalize to $[-1, 1]$ ```python p_min = np.min(sound_raw) @@ -469,9 +495,7 @@ else: sound_data = ((2 * sound_raw) - (p_min + p_max)) / (p_max - p_min) ``` -Includes a guard against division by zero when no motion is detected. - -### Step 7: Output WAV (lines 31–34, called at line 456) +### Step 7: Output WAV ```python def save_wav(samples, output_name, sample_rate): @@ -483,13 +507,15 @@ The `sample_rate` is set to the video's FPS, ensuring the output audio matches t ## 2.4 Parameters Used -| Parameter | Value | Meaning | -|-----------|-------|---------| +| Parameter | Default | Meaning | +|-----------|---------|---------| | `nlevels` | 3 | Number of wavelet decomposition scales | | `n_orient` | 6 | Number of orientations per scale (fixed by DTCWT) | | `ref_index` | 0 | Reference frame index (first frame) | | `ref_level` | 0 | Reference sub-band: finest scale | | `ref_orient` | 0 | Reference sub-band: first orientation (~$+15°$) | +| `biort` | `near_sym_b` | Biorthogonal filter for level 1 | +| `qshift` | `qshift_b` | Quarter-shift filter for levels 2+ | ## 2.5 What Each Scale Captures @@ -539,34 +565,82 @@ The vibration signal is present across all scales (the whole surface moves), but --- -# Part 4: Denoising +## Future Work + +- **Multiprocessing across frames**: Frame processing is independent after the reference frame is computed. Reading frames remains sequential (VideoCapture limitation), but the DTCWT + phase extraction can be parallelized across CPU cores using batch processing with `multiprocessing.Pool`, giving ~Nx speedup on an N-core machine. + +- **Better post-processing / signal recovery**: The current algorithm uses properly wrapped phase differences (`np.angle(coeffs * conj(ref))`, bounded to [-π, π]), which is mathematically correct but produces lower-amplitude signals for very small vibrations. Exploring better post-processing — such as phase unwrapping, adaptive Wiener filtering, or learned denoising — could recover signal strength without reintroducing the phase wrapping artifacts. + +--- + +## Development -Two denoising methods are available as post-processing, applied to the recovered audio via `--denoise`: +### Running Tests -- **Spectral subtraction** (`--denoise spectral`): Estimates a noise profile from the first ~0.1 seconds of audio (assumed to be noise-only), then subtracts it from the STFT magnitude across all frames. Simple and fast, but can introduce "musical noise" artifacts. +All tests run inside Docker — no local Python dependencies needed: -- **Morphological spectrogram filtering** (`--denoise morphological`): Converts the STFT magnitude to a grayscale image, applies binary thresholding followed by morphological erosion and dilation to create a signal/noise mask, then amplifies signal regions and attenuates noise regions. Based on the approach from [audio_denoising](https://github.com/joeljose/audio_denoising). +```bash +# CPU: lint + unit tests (builds image automatically if not found) +./test.sh -Both methods can also be used standalone on existing WAV files via `--denoise-input`: +# GPU: lint + unit tests (requires nvidia-container-toolkit) +./test.sh gpu -```sh -python visualmic.py --denoise-input sound.wav --denoise spectral -o denoised.wav -python visualmic.py --denoise-input sound.wav --denoise morphological -o denoised_morph.wav +# Force rebuild before testing +./test.sh --build +./test.sh gpu --build ``` ---- +**CPU tests** (`tests/test_visualmic.py`) cover: +- Utility functions (`format_duration`, `find_best_shift`, `save_wav`) +- Phase signal postprocessing (cross-correlation, normalization, Butterworth filter) +- VRAM estimation arithmetic +- Full `extract_audio` pipeline on synthetic 256x256 video (shape, finiteness) +- All CLI validation error paths -## Future Work +**GPU tests** (`tests/test_visualmic_gpu.py`) cover: +- DTCWTForward shapes, finiteness, and custom filter selection +- Full `extract_audio_gpu` pipeline on synthetic 256x256 video +- All GPU tests skip automatically on systems without CUDA -- **GPU-accelerated DTCWT** *(implemented)*: The `--gpu` flag uses [`pytorch_wavelets`](https://github.com/fbcotter/pytorch_wavelets) with CUDA to run batched DTCWT transforms on the GPU. See [Running with GPU acceleration](#d-running-with-gpu-acceleration-docker--cuda) for setup instructions. +### Versioning -- **Multiprocessing across frames**: Frame processing is independent after the reference frame is computed. Reading frames remains sequential (VideoCapture limitation), but the DTCWT + phase extraction can be parallelized across CPU cores using batch processing with `multiprocessing.Pool`, giving ~Nx speedup on an N-core machine. +Version is tracked in a `VERSION` file at the project root. `visualmic.py` has `__version__` baked into the source (updated at release time). + +**To cut a release:** +1. Update `VERSION` with the new version number +2. Update `__version__` in `visualmic.py` +3. Update `CHANGELOG.md` — move items from `[Unreleased]` to `[X.Y.Z] - YYYY-MM-DD` +4. Commit: `Release vX.Y.Z` +5. Tag: `git tag -a vX.Y.Z -m "Release vX.Y.Z"` +6. Push: `git push && git push origin vX.Y.Z` +7. Rebuild Docker images: `./docker-build.sh && ./docker-build-gpu.sh` -- **Better post-processing / signal recovery**: The current algorithm uses properly wrapped phase differences (`np.angle(coeffs * conj(ref))`, bounded to [-π, π]), which is mathematically correct but produces lower-amplitude signals for very small vibrations. The original (2021) implementation used unwrapped phase subtraction (`phase - ref_phase`), which could exceed [-π, π] and produced stronger (but noisier) output. Exploring better post-processing — such as phase unwrapping, adaptive Wiener filtering, or learned denoising — could recover signal strength without reintroducing the phase wrapping artifacts. +### Project Structure + +``` +visualmic.py # CLI tool (CPU + GPU paths) +Dockerfile # CPU Docker image (python:3.11-slim) +Dockerfile.gpu # GPU Docker image (pytorch:2.1.2-cuda12.1) +docker-build.sh # Build + tag CPU image +docker-build-gpu.sh # Build + tag GPU image +test.sh # Run lint + tests (Docker, supports cpu/gpu mode) +requirements.txt # CPU runtime dependencies +requirements-gpu.txt # GPU runtime dependencies +requirements-dev.txt # Dev dependencies (pytest, ruff) +tests/ + test_visualmic.py # CPU unit tests + test_visualmic_gpu.py # GPU unit tests (CUDA-only, skip on CPU) +docs/design/ # Architecture decision records + visualmic-hardening.md # Hardening design doc +VERSION # Single source of truth for version +CHANGELOG.md # Release history +CONTRIBUTING.md # Contribution guidelines +``` --- -# References +## References 1. Davis, A., Rubinstein, M., Wadhwa, N., Mysore, G.J., Durand, F., & Freeman, W.T. (2014). *The Visual Microphone: Passive Recovery of Sound from Video.* ACM Transactions on Graphics (SIGGRAPH), 33(4). [Paper PDF](https://people.csail.mit.edu/mrub/papers/VisualMic_SIGGRAPH2014.pdf) | [Project Page](https://people.csail.mit.edu/mrub/VisualMic/) @@ -598,8 +672,8 @@ python visualmic.py --denoise-input sound.wav --denoise morphological -o denoise --- ## Follow Me - - - +   +   +

Show your support by starring the repository 🙂

diff --git a/docs/design/visualmic-hardening.md b/docs/design/visualmic-hardening.md new file mode 100644 index 0000000..3914abb --- /dev/null +++ b/docs/design/visualmic-hardening.md @@ -0,0 +1,268 @@ +# Design Doc: Visual-Mic Project Hardening + +**Status: APPROVED** + +## Context + +The Visual-Mic project is a Python implementation of the Visual Microphone algorithm (Davis et al., SIGGRAPH 2014) that recovers sound from high-speed video using 2D DTCWT phase extraction. It's a single-file CLI tool (`visualmic.py`) with CPU and GPU paths, shipped via Docker. + +A code review identified gaps vs the sibling repos (EVM, DTCWT Motion Mag): no unit tests, no CI/CD, no versioning, no design docs, duplicated denoising code, outdated GPU Dockerfile, and hard-coded parameters. This design doc covers the hardening plan. + +**PRD**: https://github.com/joeljose/Visual-Mic/issues/2 + +## Goals and Non-Goals + +**Goals:** +- Remove denoising code (duplicates `joeljose/audio_denoising` repo) +- Add unit tests for CPU and GPU codepaths +- Establish versioning infrastructure (VERSION file, `__version__`, `--version`, Docker labels) +- Add `--nlevels`, `--biort`, `--qshift` CLI flags +- Add pre-flight memory/VRAM estimation +- Add CI/CD pipeline (lint + import + help + version) +- Add changelog, contributing guide, design doc, dev dependencies +- Upgrade GPU Dockerfile to PyTorch 2.1.2 / CUDA 12.1 +- Clean up README (remove denoising, add Development section, restructure Setup) + +**Non-Goals:** +- Refactoring GPU/CPU into shared code paths (defer — `postprocess_phase_signals` already shared) +- Multiprocessing across frames (defer) +- Refactoring into a pip-installable package +- GPU testing in CI (no hosted GPU runners) +- Integration tests with audio output comparison + +## Proposed Design + +### A. Remove Denoising Code + +Delete from `visualmic.py`: +- `denoise_spectral()` (lines 37–57) and `denoise_morphological()` (lines 60–86) +- `--denoise` and `--denoise-input` CLI arguments +- Standalone denoise mode block (lines 321–340) +- Post-pipeline denoise application (lines 448–454) +- `from scipy.io.wavfile import read as read_wav` (only needed for `--denoise-input`) +- `from scipy import ndimage` (only needed for morphological denoising) +- Update error message at line 344 to remove `--denoise-input` reference + +The `audio_denoising` repo at `github.com/joeljose/audio_denoising` provides this functionality separately. + +### B. CLI Flag Additions + +**`--nlevels` (default: 3):** +Currently hard-coded at line 419. Make it a CLI argument. Validation: must be >= 1. Update `min_dim = 2 ** nlevels` (line 420) to use the argument. + +**`--biort` (default: `near_sym_b`) and `--qshift` (default: `qshift_b`):** +Matching DTCWT Motion Mag v2.0.0 defaults. Applied to both CPU and GPU paths: +- CPU: `transform.forward(gray, nlevels=nlevels, biort=biort, qshift=qshift)` — note: `dtcwt.Transform2d()` accepts biort/qshift in constructor +- GPU: `DTCWTForward(J=nlevels, biort=biort, qshift=qshift)` — already parameterized in the code (line 202), just needs to read from args instead of hard-coding + +Available options (same as DTCWT Motion Mag): +- biort: `antonini`, `legall`, `near_sym_a`, `near_sym_b` +- qshift: `qshift_06`, `qshift_a`, `qshift_b`, `qshift_c`, `qshift_d` + +**`--version`:** +Standard argparse version action reading from `__version__`. + +### C. Pre-Flight Memory Estimation + +**CPU path:** +Visual-Mic is memory-efficient — it streams frames and only stores the phase signal array `(num_frames, nlevels, n_orient)` which is small (e.g., 301 × 3 × 6 × 8 bytes = 43 KB). The main memory consumer is the reference conjugate coefficients (one frame's worth of DTCWT coefficients). No pre-flight check needed for CPU — it can handle arbitrarily long videos. + +**GPU path:** +The GPU path batches frames, so peak VRAM depends on batch size: +``` +vram_per_frame = H × W × 4 bytes (float32 input) +dtcwt_overhead = ~15× per frame (forward transform intermediates) +vram_estimate = batch_size × H × W × 4 × 15 + 300 MB (PyTorch overhead) +``` + +Add `estimate_vram()` function. Before processing, query `torch.cuda.mem_get_info()`, compare against estimate, warn if > 70% of available VRAM. Suggest reducing `--batch-size` if insufficient. + +Also add OOM error handling in the GPU batch loop (already exists at line 236, just needs consistent messaging matching the other repos): +``` +Error: GPU out of memory with batch_size={batch_size}. + Suggestions: + - Reduce --batch-size (current: {batch_size}) + - Use --roi to crop to a smaller region + - Remove --gpu to use CPU mode +``` + +### D. Versioning + +Same pattern as EVM and DTCWT Motion Mag: +- `VERSION` file at repo root containing `2.0.0` +- `__version__ = "2.0.0"` baked into `visualmic.py` +- `--version` CLI flag via argparse +- Docker build scripts read `VERSION`, tag images as `visual-mic:{version}` + `visual-mic:latest` +- Dockerfiles receive `--build-arg VERSION` and apply `LABEL version=${VERSION}` + +### E. Testing Infrastructure + +**`requirements-dev.txt`:** +``` +pytest>=7.0,<9 +ruff>=0.4.0,<1 +``` + +**`tests/test_visualmic.py`** — CPU tests: + +Tier 1 (strict, exact equality): +- `TestFormatDuration`: 0s, 59s, 60s, 3661s +- `TestFindBestShift`: known shift on synthetic signals (sine wave shifted by N samples) +- `TestSaveWav`: output file exists, correct sample rate, int16 range + +Tier 2 (moderate tolerance): +- `TestPostprocessPhaseSignals`: constant input → silent output, normalization to [-1, 1], cross-correlation alignment on known-shifted synthetic signals +- `TestButterworthFilter`: in-band signal preserved, out-of-band signal attenuated, freq_low >= Nyquist warning +- `TestEstimateVram`: arithmetic correctness, batch size scaling + +Tier 3 (smoke): +- `TestExtractAudio`: synthetic 256×256 video (random noise, 32 frames), verify output shape = (32,), all values finite, values in [-1, 1] +- `TestInputValidation`: missing file, invalid frequencies, bad ROI format, ROI out of bounds, invalid nlevels + +**`tests/test_visualmic_gpu.py`** — GPU tests: + +All wrapped in `@pytest.mark.skipif(not HAS_CUDA)`: +- `TestGpuForwardPass`: DTCWTForward on 256×256 batch, verify Yh shapes and finite values +- `TestExtractAudioGpu`: smoke test on synthetic 256×256 video, verify output shape and finiteness +- `TestEstimateVram`: VRAM arithmetic + +**`test.sh`:** +Matches DTCWT Motion Mag pattern — Docker-based, supports `cpu`/`gpu` modes: +```bash +./test.sh # CPU lint + tests +./test.sh gpu # GPU lint + tests +./test.sh --build # Force rebuild +``` + +### F. CI/CD + +**`.github/workflows/ci.yml`:** +Two jobs matching EVM/DTCWT pattern: + +CPU job: +1. Build Docker image (CPU Dockerfile) +2. Lint (`ruff check .`) +3. Verify import (`python -c "import visualmic"`) +4. Verify `--help` +5. Verify `--version` + +GPU job: +1. Build Docker image (GPU Dockerfile) +2. Lint (`ruff check .`) +3. Verify import +4. Verify `--help` +5. Verify `--version` + +No pipeline smoke test in CI (no test video committed to repo — MIT CSAIL videos are 14 GB). + +**Manual GPU verification (post-implementation):** +Download a MIT CSAIL test video (e.g., `Chips1-2200Hz-Mary_Had-input.avi` from http://data.csail.mit.edu/vidmag/VisualMic/Results/) and run the GPU pipeline end-to-end to verify audio output is recognizable. CPU verification with real videos is impractical — the MIT CSAIL videos are 704×704 × 22,859 frames, which would take too long or OOM on CPU. + +### G. GPU Dockerfile Upgrade + +**From:** `pytorch/pytorch:1.12.1-cuda11.3-cudnn8-runtime` +**To:** `pytorch/pytorch:2.1.2-cuda12.1-cudnn8-runtime` + +**Verified compatible:** Live-tested `pytorch_wavelets.DTCWTForward` with same API Visual-Mic uses (J=3, biort='near_sym_b', qshift='qshift_b', Yh[level][..., 0/1] indexing) on PyTorch 2.1.2 / CUDA 12.1. All outputs finite and correct shapes. + +**`requirements-gpu.txt` changes:** +``` +scipy>=1.7.1,<2 +numpy>=1.20.3,<2 # <2 required: pytorch_wavelets uses removed NumPy 2.0 APIs +opencv-python-headless>=4.7.0,<5 +PyWavelets>=1.1.0 +``` + +Both Dockerfiles updated to: +- Copy `tests/` and `requirements-dev.txt` +- Accept `--build-arg VERSION` and apply `LABEL version=${VERSION}` + +### H. Documentation + +**`CHANGELOG.md`:** +Keep a Changelog format, starting at v2.0.0 (no backfill): +```markdown +# Changelog + +## [2.0.0] - YYYY-MM-DD +### Removed +- Denoising (`--denoise`, `--denoise-input`). Use `audio_denoising` repo instead. + +### Added +- `--nlevels`, `--biort`, `--qshift` CLI flags +- `--version` flag +- Pre-flight GPU memory estimation +- Unit tests (CPU + GPU) +- CI/CD pipeline +- CHANGELOG.md, CONTRIBUTING.md, design docs + +### Changed +- Default wavelet filters: `near_sym_b`/`qshift_b` (both CPU and GPU paths) +- GPU Dockerfile upgraded to PyTorch 2.1.2 / CUDA 12.1 +- Docker images now tagged with version numbers +``` + +**`CONTRIBUTING.md`:** +Same structure as EVM/DTCWT repos: +- Open issue first +- Fork and branch from main +- Small PRs, one logical change +- PEP 8 + ruff +- Run `./test.sh` before opening PR + +**README changes:** +- Remove Part 4 (Denoising) entirely +- Restructure Setup section A (remove YouTube link, match terse style of other repos) +- Add CI badge at top +- Add Development section (Running Tests, Versioning, Project Structure) +- Update CLI flags table with `--nlevels`, `--biort`, `--qshift`, `--version` +- Update Future Work: remove "GPU-accelerated DTCWT" (done), remove denoising reference +- Add project structure tree + +## Alternatives Considered + +### Remove denoising vs keep as optional + +| Approach | Pros | Cons | Verdict | +|----------|------|------|---------| +| Remove entirely | Single responsibility, less code to maintain, separate repo exists | Users lose integrated denoising | **Chosen** — `audio_denoising` repo is the right place for this | +| Keep as optional | Convenience for users | Duplicated code, scope creep, adds scipy.ndimage dependency | Rejected | + +### Default wavelet filters: near_sym_a vs near_sym_b + +| Approach | Pros | Cons | Verdict | +|----------|------|------|---------| +| Keep `near_sym_a` (dtcwt default) | Backward compatible | Inconsistent with GPU path (already uses `near_sym_b`), worse quality at high magnification | Rejected | +| Switch to `near_sym_b` | Matches DTCWT Motion Mag, matches existing GPU path, fewer artifacts | Breaking change for CPU users | **Chosen** — v2.0.0 justifies the break, `--biort`/`--qshift` give user control | + +### Memory estimation: CPU + GPU vs GPU only + +| Approach | Pros | Cons | Verdict | +|----------|------|------|---------| +| Both CPU and GPU | Consistent with DTCWT Motion Mag | CPU path is streaming (tiny memory footprint), check is pointless | Rejected — would always pass | +| GPU only | Targets actual OOM risk | No CPU protection | **Chosen** — CPU path streams frames, only stores phase_signals array (~KB), OOM is not a realistic risk | + +### Version: 1.1.0 vs 2.0.0 + +| Approach | Pros | Cons | Verdict | +|----------|------|------|---------| +| 1.1.0 | Lower number | Violates semver — removing CLI flags is breaking | Rejected | +| 2.0.0 | Semver correct, matches DTCWT precedent | Higher number | **Chosen** — breaking changes require major bump | + +## Tradeoffs and Risks + +- **`pytorch_wavelets` is unmaintained** (last commit 2023). Uses stable PyTorch APIs (`F.conv2d`, `autograd.Function`), but `pkg_resources` will break on Python 3.14+. Mitigated: pin PyTorch 2.1.2 and numpy<2 in Docker. Can fork if needed. + +- **CPU/GPU outputs differ.** Different DTCWT implementations (`dtcwt` vs `pytorch_wavelets`), different precision (float64 vs float32). Both produce valid audio; they are not cross-comparable. Documented, accepted. + +- **No pipeline smoke test in CI.** Unlike EVM/DTCWT repos (which have `face.mp4` committed), Visual-Mic has no small test video in the repo. Mitigated: synthetic video tests run locally via `./test.sh`, and manual GPU verification against MIT CSAIL videos post-implementation. + +- **CPU path untested with real videos.** MIT CSAIL videos (704×704, 22K+ frames) are too large for CPU processing in reasonable time or memory. GPU path is the only practical way to verify against real data. + +- **GPU memory estimation is approximate.** PyTorch's actual VRAM usage depends on allocator behavior, fragmentation, and cuDNN workspace sizes. The 70% threshold provides safety margin. + +- **Default filter change is breaking for CPU path.** CPU path previously used `dtcwt` defaults (`near_sym_a`/`qshift_a`), now explicitly uses `near_sym_b`/`qshift_b`. Users can restore old behavior with `--biort near_sym_a --qshift qshift_a`. + +## Open Questions + +None — all decisions resolved during PRD and grill phases.