1"""Channel vocoder: make a synth chord speak with a recording's voice. 2 3The speech is split into log-spaced bands; each band's loudness envelope 4drives the same band of a sawtooth chord. Each carrier band is normalised 5to unit RMS first, so the envelope alone sets its level (a raw sawtooth's 61/f spectrum would otherwise bury the consonant bands). Bands above 7--noise-above also get white noise in the carrier, which is what makes 8s, t and k audible: a chord has no energy that looks like a hiss. 9 10 vocode speech.wav out.wav --chord "a2 a3 c4 e4" 11""" 12 13import argparse 14 15import numpy as np 16import scipy.io.wavfile as wavfile 17import scipy.signal as sg 18 19PITCH_CLASS = { 20 "c": 0, "c#": 1, "db": 1, "d": 2, "d#": 3, "eb": 3, "e": 4, "f": 5, 21 "f#": 6, "gb": 6, "g": 7, "g#": 8, "ab": 8, "a": 9, "a#": 10, "bb": 10, 22 "b": 11, 23} 24 25 26def note_hz(name): 27 """'a3' -> 220.0, 'c#4' -> 277.18 (scientific pitch, A4 = 440).""" 28 name = name.lower() 29 pitch, octave = name[:-1], int(name[-1]) 30 midi = 12 * (octave + 1) + PITCH_CLASS[pitch] 31 return 440.0 * 2 ** ((midi - 69) / 12) 32 33 34def read_mono(path): 35 sr, x = wavfile.read(path) 36 x = x.astype(np.float64) 37 if x.ndim > 1: 38 x = x.mean(axis=1) 39 peak = np.abs(x).max() 40 return sr, x / peak if peak > 0 else x 41 42 43def carrier(freqs, n, sr, rng): 44 """Sawtooth chord, each note doubled slightly detuned for width.""" 45 t = np.arange(n) / sr 46 out = np.zeros(n) 47 for f in freqs: 48 for detune in (1.0, 1.003): 49 phase = rng.uniform(0, 2 * np.pi) 50 out += sg.sawtooth(2 * np.pi * f * detune * t + phase) 51 return out / (2 * len(freqs)) 52 53 54def vocode(x, sr, freqs, bands, noise, noise_above, dry, env_hz, seed): 55 rng = np.random.default_rng(seed) 56 car = carrier(freqs, len(x), sr, rng) 57 hiss = rng.standard_normal(len(x)) * noise 58 top = min(9000.0, 0.45 * sr) 59 edges = np.geomspace(80.0, top, bands + 1) 60 env_lp = sg.butter(2, env_hz, fs=sr, output="sos") 61 out = np.zeros(len(x)) 62 for lo, hi in zip(edges[:-1], edges[1:]): 63 band = sg.butter(4, [lo, hi], btype="band", fs=sr, output="sos") 64 env = np.clip(sg.sosfiltfilt(env_lp, np.abs(sg.sosfilt(band, x))), 0, None) 65 source = car + (hiss if lo >= noise_above else 0) 66 c = sg.sosfilt(band, source) 67 rms = np.sqrt(np.mean(c ** 2)) 68 if rms > 0: 69 out += c / rms * env 70 out += dry * x 71 peak = np.abs(out).max() 72 return out / peak * 0.9 if peak > 0 else out 73 74 75def main(): 76 ap = argparse.ArgumentParser(description=__doc__.split("\n")[0]) 77 ap.add_argument("speech", help="input speech, WAV") 78 ap.add_argument("out", help="output WAV") 79 ap.add_argument("--chord", required=True, 80 help='notes, space-separated, e.g. "a2 a3 c4 e4"') 81 ap.add_argument("--bands", type=int, default=28) 82 ap.add_argument("--noise", type=float, default=0.3, 83 help="level of the white-noise carrier in high bands") 84 ap.add_argument("--noise-above", type=float, default=2500.0, 85 help="Hz above which bands get the noise carrier") 86 ap.add_argument("--dry", type=float, default=0.0, 87 help="amount of unprocessed speech mixed back in") 88 ap.add_argument("--env-hz", type=float, default=30.0, 89 help="envelope smoothing cutoff; lower is smoother") 90 ap.add_argument("--seed", type=int, default=0) 91 a = ap.parse_args() 92 93 sr, x = read_mono(a.speech) 94 freqs = [note_hz(n) for n in a.chord.split()] 95 y = vocode(x, sr, freqs, a.bands, a.noise, a.noise_above, a.dry, 96 a.env_hz, a.seed) 97 wavfile.write(a.out, sr, (y * 32767).astype(np.int16)) 98 99 100if __name__ == "__main__": 101 main()