jevstrudel.git / tools / vocals / vocode.py
1"""Channel vocoder: make a synth chord speak with a recording's voice.
2
3The speech is split into log-spaced bands; each band's loudness envelope
4drives the same band of a sawtooth chord. Each carrier band is normalised
5to unit RMS first, so the envelope alone sets its level (a raw sawtooth's
61/f spectrum would otherwise bury the consonant bands). Bands above
7--noise-above also get white noise in the carrier, which is what makes
8s, t and k audible: a chord has no energy that looks like a hiss.
9
10    vocode speech.wav out.wav --chord "a2 a3 c4 e4"
11"""
12
13import argparse
14
15import numpy as np
16import scipy.io.wavfile as wavfile
17import scipy.signal as sg
18
19PITCH_CLASS = {
20    "c": 0, "c#": 1, "db": 1, "d": 2, "d#": 3, "eb": 3, "e": 4, "f": 5,
21    "f#": 6, "gb": 6, "g": 7, "g#": 8, "ab": 8, "a": 9, "a#": 10, "bb": 10,
22    "b": 11,
23}
24
25
26def note_hz(name):
27    """'a3' -> 220.0, 'c#4' -> 277.18 (scientific pitch, A4 = 440)."""
28    name = name.lower()
29    pitch, octave = name[:-1], int(name[-1])
30    midi = 12 * (octave + 1) + PITCH_CLASS[pitch]
31    return 440.0 * 2 ** ((midi - 69) / 12)
32
33
34def read_mono(path):
35    sr, x = wavfile.read(path)
36    x = x.astype(np.float64)
37    if x.ndim > 1:
38        x = x.mean(axis=1)
39    peak = np.abs(x).max()
40    return sr, x / peak if peak > 0 else x
41
42
43def carrier(freqs, n, sr, rng):
44    """Sawtooth chord, each note doubled slightly detuned for width."""
45    t = np.arange(n) / sr
46    out = np.zeros(n)
47    for f in freqs:
48        for detune in (1.0, 1.003):
49            phase = rng.uniform(0, 2 * np.pi)
50            out += sg.sawtooth(2 * np.pi * f * detune * t + phase)
51    return out / (2 * len(freqs))
52
53
54def vocode(x, sr, freqs, bands, noise, noise_above, dry, env_hz, seed):
55    rng = np.random.default_rng(seed)
56    car = carrier(freqs, len(x), sr, rng)
57    hiss = rng.standard_normal(len(x)) * noise
58    top = min(9000.0, 0.45 * sr)
59    edges = np.geomspace(80.0, top, bands + 1)
60    env_lp = sg.butter(2, env_hz, fs=sr, output="sos")
61    out = np.zeros(len(x))
62    for lo, hi in zip(edges[:-1], edges[1:]):
63        band = sg.butter(4, [lo, hi], btype="band", fs=sr, output="sos")
64        env = np.clip(sg.sosfiltfilt(env_lp, np.abs(sg.sosfilt(band, x))), 0, None)
65        source = car + (hiss if lo >= noise_above else 0)
66        c = sg.sosfilt(band, source)
67        rms = np.sqrt(np.mean(c ** 2))
68        if rms > 0:
69            out += c / rms * env
70    out += dry * x
71    peak = np.abs(out).max()
72    return out / peak * 0.9 if peak > 0 else out
73
74
75def main():
76    ap = argparse.ArgumentParser(description=__doc__.split("\n")[0])
77    ap.add_argument("speech", help="input speech, WAV")
78    ap.add_argument("out", help="output WAV")
79    ap.add_argument("--chord", required=True,
80                    help='notes, space-separated, e.g. "a2 a3 c4 e4"')
81    ap.add_argument("--bands", type=int, default=28)
82    ap.add_argument("--noise", type=float, default=0.3,
83                    help="level of the white-noise carrier in high bands")
84    ap.add_argument("--noise-above", type=float, default=2500.0,
85                    help="Hz above which bands get the noise carrier")
86    ap.add_argument("--dry", type=float, default=0.0,
87                    help="amount of unprocessed speech mixed back in")
88    ap.add_argument("--env-hz", type=float, default=30.0,
89                    help="envelope smoothing cutoff; lower is smoother")
90    ap.add_argument("--seed", type=int, default=0)
91    a = ap.parse_args()
92
93    sr, x = read_mono(a.speech)
94    freqs = [note_hz(n) for n in a.chord.split()]
95    y = vocode(x, sr, freqs, a.bands, a.noise, a.noise_above, a.dry,
96               a.env_hz, a.seed)
97    wavfile.write(a.out, sr, (y * 32767).astype(np.int16))
98
99
100if __name__ == "__main__":
101    main()