zefie icon

restore mono segments to stereo

zefie | PRO | 11/23/24 02:05:21 AM UTC | 0 ⭐ | 695 👁️ | Never ⏰ | []
Python |

6.36 KB

|

Source Code

|

0 👍

/

0 👎

import librosa
import numpy as np
import soundfile as sf
 
# Let's be real. Most of this was generated by AI.
 
# Load your stereo audio file
y_stereo, sr = librosa.load('/mnt/d/instrumentals/motm.wav', sr=None, mono=False, dtype=np.float32)
 
# Ensure the audio is stereo
if y_stereo.ndim != 2 or y_stereo.shape[0] != 2:
    raise ValueError("Input file must be a stereo file.")
 
coh_threshold = 0.85  # High coherence indicates mono content
# Define the threshold for determining mono sections
energy_diff_threshold = 0.15  # Low energy difference indicates mono content
 
# Initialize an array to store the indices of mono sections
frame_length = 2048
hop_length = frame_length // 2
mono_indices = []
 
def is_mono_segment(y_left, y_right, silence_threshold=1e-5):
 
    """
    Determine if a segment is mono based on coherence and energy difference.
    
    Parameters:
        y_left (numpy.ndarray): Left channel segment.
        y_right (numpy.ndarray): Right channel segment.
        coh_threshold (float): Threshold for coherence.
        energy_diff_threshold (float): Threshold for energy difference.
        silence_threshold (float): Minimum energy to detect non-silent segments.
        
    Returns:
        bool: True if the segment is mono, False otherwise.
    """
    # Check if the segment is silent (skip silent frames)
    if np.sum(y_left ** 2) + np.sum(y_right ** 2) < silence_threshold:
        return False
 
    # Compute coherence
    coh = np.corrcoef(y_left, y_right)[0, 1]
 
    # Compute energy difference
    diff_energy = np.sum((y_left - y_right) ** 2) / (np.sum(y_left ** 2) + np.sum(y_right ** 2) + 1e-10)
 
    # Determine if the segment is mono
    return coh > coh_threshold and diff_energy < energy_diff_threshold
 
 
def add_stereo_delay(mono_signal, delay_ms, sr):
    """
    Simulate stereo by introducing a small delay to one channel.
    
    Parameters:
        mono_signal (numpy.ndarray): The mono signal.
        delay_ms (float): Delay in milliseconds.
        sr (int): Sampling rate of the audio.
    
    Returns:
        numpy.ndarray: Stereo signal with simulated stereo effect.
    """
    # Convert delay from milliseconds to samples
    delay_samples = int((delay_ms / 1000) * sr)
 
    # Create a delayed version of the signal
    delayed_signal = np.pad(mono_signal, (delay_samples, 0))[:len(mono_signal)]
 
    # Combine into stereo: left channel is original, right channel is delayed
    stereo_signal = np.column_stack((mono_signal, delayed_signal))
 
    return stereo_signal
 
def simulate_stereo_with_phase_shift(mono_signal, phase_shift_degrees, sr):
    """
    Simulate stereo by introducing a phase shift between channels.
    
    Parameters:
        mono_signal (numpy.ndarray): The mono signal.
        phase_shift_degrees (float): Phase shift in degrees.
        sr (int): Sampling rate of the audio.
    
    Returns:
        numpy.ndarray: Stereo signal with simulated stereo effect.
    """
    phase_shift_samples = int((phase_shift_degrees / 360) * sr)
    shifted_signal = np.roll(mono_signal, phase_shift_samples)
    reverb_signal = librosa.effects.preemphasis(shifted_signal) * 0.2
 
    # Combine into stereo
    stereo_signal = np.column_stack((mono_signal, shifted_signal + reverb_signal))
    
    #return np.column_stack((mono_signal, shifted_signal))
    return stereo_signal
 
def calculate_phase_shift(y_stereo, sr, frame_length=2048, hop_length=1024):
    """
    Calculate the phase shift between the left and right channels in a stereo signal.
 
    Parameters:
        y_stereo (numpy.ndarray): Stereo audio signal (2, n_samples).
        sr (int): Sampling rate of the audio.
        frame_length (int): STFT frame length.
        hop_length (int): STFT hop length.
 
    Returns:
        float: Average phase shift in degrees between the left and right channels.
    """
    # Extract the left and right channels
    y_left = y_stereo[0]
    y_right = y_stereo[1]
 
    # Compute STFT for both channels
    S_left = librosa.stft(y_left, n_fft=frame_length, hop_length=hop_length)
    S_right = librosa.stft(y_right, n_fft=frame_length, hop_length=hop_length)
 
    # Compute phase difference between left and right channels
    phase_left = np.angle(S_left)
    phase_right = np.angle(S_right)
    phase_diff = phase_right - phase_left  # Phase difference in radians
 
    # Unwrap the phase to avoid discontinuities
    phase_diff_unwrapped = np.unwrap(phase_diff, axis=0)
 
    # Calculate the mean phase difference across frequencies and frames
    mean_phase_diff = np.mean(phase_diff_unwrapped)
 
    # Convert to degrees
    phase_shift_degrees = np.degrees(mean_phase_diff)
 
    return phase_shift_degrees
 
# Function to restore mono sections
def restore_mono_sections(signal, indices, frame_length, hop_length):
    restored_signal = np.copy(signal)
    GAIN_FACTOR = 1.25 
    for t in indices:
        y_left = signal[0, t:t+frame_length]
        y_right = signal[1, t:t+frame_length]
 
        # Use the average of the channels to create a mono section
        mono_frame = (y_left + y_right) / 2
 
        # Generate simulated stereo for the mono frame
        stereo_frame = simulate_stereo_with_phase_shift(mono_frame, 270, sr)
 
        stereo_frame *= GAIN_FACTOR
        
        # Assign the stereo frame to the correct slices of the `restored_signal`
        restored_signal[0, t:t+frame_length] = stereo_frame[:, 0]  # Left channel
        restored_signal[1, t:t+frame_length] = stereo_frame[:, 1]  # Right channel
 
    return restored_signal
    
mono_indices = []
 
import scipy.signal
 
# Smooth mono detection with a moving average filter
frame_status = np.zeros(y_stereo.shape[1] // hop_length)
for t in range(len(frame_status)):
    start = t * hop_length
    end = start + frame_length
    if is_mono_segment(y_stereo[0, start:end], y_stereo[1, start:end]):
        frame_status[t] = 1
 
# Smooth the binary decision with a moving average
smoothed_status = scipy.signal.convolve(frame_status, np.ones(5) / 5, mode='same')
 
# Identify smoothed mono indices
mono_indices = np.where(smoothed_status > 0.5)[0] * hop_length
 
# Restore the audio
y_restored_stereo = restore_mono_sections(y_stereo, mono_indices, frame_length, hop_length)
 
# Save your restored audio file
sf.write('/mnt/d/instrumentals/motm_test.wav', y_restored_stereo.T, sr)
 

Comments