import numpy as np import scipy.signal as signal import sounddevice as sd import time # --- CONFIGURATION --- FS = 16000 # 16kHz Sample Rate FRAME_MS = 25 # 25ms Frame size FRAME_LEN = int(FS * FRAME_MS / 1000) DURATION = 3 # 3 Seconds recording BITS_PER_FRAME = 12 # Target payload def v_uv_discriminator(frame): """ Finds Voiced/Unvoiced state and F0 (Pitch) using Time-Lag Autocorrelation. This prevents "Pitch Doubling/Tripling" by tracking the physical glottal impulse distance rather than the First Formant (F1). """ energy = np.mean(frame**2) if energy < 1e-6: return False, 0.0, energy # 1. LOW PASS FILTER (Isolate the low-end rumble) b, a = signal.butter(2, 600 / (FS / 2), btype='low') lpf_frame = signal.lfilter(b, a, frame) # 2. AUTOCORRELATION corr = np.correlate(lpf_frame, lpf_frame, mode='full') corr = corr[len(corr)//2:] # Keep only the positive lags (0 to end) # Normalize corr = corr / (corr[0] + 1e-9) # 3. SEARCH FOR THE GLOTTAL PULSE (Human Pitch Range) # A human pitch of 50 Hz to 400 Hz corresponds to specific array indices (lags) min_lag = int(FS / 400) # Max pitch ~ 400 Hz (Index 40 at 16kHz) max_lag = int(FS / 50) # Min pitch ~ 50 Hz (Index 320 at 16kHz) if max_lag >= len(corr): return False, 0.0, energy # Find the highest peak in the valid human pitch range valid_corr = corr[min_lag:max_lag] peak_idx = np.argmax(valid_corr) # Calculate the true lag in the original array true_lag = min_lag + peak_idx # 4. VOICED / UNVOICED DECISION # If the peak is strong, it's a periodic vowel. If it's weak/mushy, it's an "S" or "F". periodicity = corr[true_lag] if periodicity > 0.45: # 0.45 is a great threshold for human speech f0 = FS / true_lag return True, f0, energy else: return False, 0.0, energy def extract_poles_spline(frame, is_voiced, order=12): """ Extracts Spectral Envelope poles using the Annihilating Filter approach. Uses a least-squares solver to find the filter coefficients that minimize the residual. """ if not is_voiced: # Calculate Autocorrelation to find deterministic envelope of noise corr = np.correlate(frame, frame, mode='full') proc_signal = corr[len(corr)//2 : len(corr)//2 + FRAME_LEN//2] else: proc_signal = frame n = len(proc_signal) if n <= order: return np.array([1.0] + [0.0] * order) # Build the observation matrix y = proc_signal[order:] X = np.zeros((n - order, order)) for i in range(order): X[:, i] = proc_signal[order - i - 1 : n - i - 1] # Solve for a_coeffs using Least Squares (Algebraic approach) try: a_coeffs, _, _, _ = np.linalg.lstsq(X, -y, rcond=None) a = np.concatenate(([1.0], a_coeffs)) # Stability check roots = np.roots(a) if np.any(np.abs(roots) > 1.0): roots = roots / (np.abs(roots) + 1e-6) a = np.poly(roots).real # === BANDWIDTH EXPANSION === # Heavily dampen high-order 'plastic' ringing gamma = 0.95 for k in range(len(a)): a[k] = a[k] * (gamma ** k) return a except np.linalg.LinAlgError: return np.array([1.0] + [0.0] * order) def synthesize_frame(f0, energy, poles, is_voiced, zi, phase_offset): """ Continuous-Time Synthesis with Mixed Excitation. Kills the 'plastic foil' effect by smoothing the glottal pulse and adding natural human breathiness to vowels. """ if is_voiced and f0 > 50: period = int(FS / f0) source = np.zeros(FRAME_LEN) # 1. CONTINUOUS PHASE indices = np.arange(phase_offset, FRAME_LEN, period) if len(indices) > 0: source[indices] = 1.0 # Calculate where the first pulse of the NEXT frame belongs phase_offset = indices[-1] + period - FRAME_LEN else: phase_offset = max(0, phase_offset - FRAME_LEN) # 2. GLOTTAL ROLL-OFF (Kill the high-frequency plastic buzz) # Real vocal cords drop off sharply above 3kHz. b_glot, a_glot = signal.butter(2, 3000 / (FS / 2), btype='low') source = signal.lfilter(b_glot, a_glot, source) # 3. MIXED EXCITATION (Restore human breathiness) # Pure impulses sound sterile. We add 15% noise to make it warm. breath = np.random.normal(0, 1, FRAME_LEN) * 0.15 source += breath source *= np.sqrt(energy / (np.mean(source**2) + 1e-9)) * 2.5 else: source = np.random.normal(0, 1, FRAME_LEN) source *= np.sqrt(energy / (np.mean(source**2) + 1e-9)) * 0.3 # Randomize phase for unvoiced so it doesn't accidentally sync up phase_offset = 0 # FILTER MEMORY: Pass zi into the filter, and get the new zf out # If the lengths mismatch (due to order changes), reset zi if len(zi) != len(poles) - 1: zi = np.zeros(len(poles) - 1) synth, zf = signal.lfilter([1.0], poles, source, zi=zi) return synth, zf, phase_offset def main(): print(f"--- Blackout Comms: Spline Vocoder Prototype V7 ---") try: print(f"Recording {DURATION} seconds of audio at {FS}Hz...") audio = sd.rec(int(DURATION * FS), samplerate=FS, channels=1) sd.wait() audio = audio.flatten() print("Recording complete. Processing...") except Exception as e: print(f"Microphone not available ({e}). Generating synthetic test audio...") t = np.linspace(0, DURATION, int(FS * DURATION)) audio = 0.5 * np.sin(2 * np.pi * 150 * t) + 0.2 * np.sin(2 * np.pi * 300 * t) audio += 0.1 * np.random.normal(0, 1, len(t)) # Pre-emphasis audio = np.append(audio[0], audio[1:] - 0.97 * audio[:-1]) encoded_payload = [] reconstructed_audio = np.array([]) # Initialize Memory for Continuous Physics ORDER = 12 filter_memory = np.zeros(ORDER) glottal_phase = 0 # ENCODE / DECODE LOOP for i in range(0, len(audio) - FRAME_LEN, FRAME_LEN): frame = audio[i : i + FRAME_LEN] # 1. Pitch Tracking (Uses raw frame, internal LPF handles edges) is_voiced, f0, energy = v_uv_discriminator(frame) # === THE CELLOPHANE FIX: WINDOWING === # Taper the edges of the frame to 0 so the Spline Operator # doesn't try to encode the sharp, high-frequency frame cuts. windowed_frame = frame * np.hamming(FRAME_LEN) # 2. Extract Poles (From the WINDOWED frame) poles = extract_poles_spline(windowed_frame, is_voiced, order=ORDER) encoded_payload.append({ 'v': is_voiced, 'f': f0 if is_voiced else energy, 'p': poles }) # 3. Synthesize synth_frame, filter_memory, glottal_phase = synthesize_frame( f0, energy, poles, is_voiced, filter_memory, glottal_phase ) reconstructed_audio = np.append(reconstructed_audio, synth_frame) print(f"Transmission Simulated: {len(encoded_payload)} frames @ ~48 bits/frame = {len(encoded_payload)*48/DURATION:.1f} bps") # Post-processing (De-emphasis) reconstructed_audio = signal.lfilter([1.0], [1.0, -0.97], reconstructed_audio) print("Playing reconstructed audio...") sd.play(reconstructed_audio, FS) sd.wait() print("Done.") if __name__ == "__main__": main()