#!/usr/bin/env bash # Reproduces the Pokedex voice spike end to end: # TTS (espeak-ng and Piper) -> robot-voice DSP filter -> mp3 # # Setup (Debian/Ubuntu): # sudo apt-get install -y espeak-ng ffmpeg # pip install piper-tts --break-system-packages # pip install numpy scipy --break-system-packages # # Voice model: # Official route: download a Piper voice (.onnx + .onnx.json) from the # Piper voices repo and point --model at it, e.g. en_US-lessac-medium. # In this sandbox, huggingface.co wasn't reachable, so I instead pulled a # community PyPI wheel that bundles the model file directly: # pip download --no-deps joe-us-piper-voice # (unzip the wheel; the .onnx/.onnx.json live under # joe_us_piper_voice/data/). Not an official source -- for production, # get voices from Piper's own releases instead. set -euo pipefail cd "$(dirname "$0")" VOICE_MODEL="voices/en_US-joe-medium.onnx" export ALSA_CONFIG_PATH=/dev/null # silence ALSA warnings in headless envs # ---- Stage 1: raw espeak-ng (formant synth, offline, inherently "robotic") ---- espeak-ng -v en-us -s 150 -w bulbasaur_raw.wav \ "Bulbasaur. Seed pokemon. It can be seen napping in bright sunlight. There is a seed on its back. By soaking up the sun's rays, the seed grows progressively larger." espeak-ng -v en-us -s 150 -w charizard_raw.wav \ "Charizard. Flame pokemon. Charizard flies around the sky in search of powerful opponents. It breathes fire of such great heat that it melts anything." espeak-ng -v en-us -s 150 -w pikachu_raw.wav \ "Pikachu. Mouse pokemon. When several of these pokemon gather, their electricity could build and cause lightning storms." # ---- Stage 2: Piper neural TTS, first pass (natural but bad cadence on rare words) ---- gen_neural() { local name="$1" text="$2" echo "$text" | piper -m "$VOICE_MODEL" -f "${name}_neural.wav" } gen_neural bulbasaur "Bulbasaur. Seed pokemon. It can be seen napping in bright sunlight. There is a seed on its back. By soaking up the sun's rays, the seed grows progressively larger." gen_neural charizard "Charizard. Flame pokemon. Charizard flies around the sky in search of powerful opponents. It breathes fire of such great heat that it melts anything." gen_neural pikachu "Pikachu. Mouse pokemon. When several of these pokemon gather, their electricity could build and cause lightning storms." # ---- Stage 3: Piper neural TTS, cadence-fixed pass ---- # Fix = (a) standard "X, the Y Pokemon." phrasing instead of two short # sentences, which stopped "Pokemon" from landing phrase-final where # duration models over-lengthen it, and (b) reduced noise-w-scale (duration # randomness) and noise-scale (audio variance) so rare proper nouns don't # get a random stretched/warped rendering. gen_fixed() { local name="$1" text="$2" echo "$text" | piper -m "$VOICE_MODEL" -f "${name}_fixed.wav" \ --noise-scale 0.5 --noise-w-scale 0.3 --length-scale 0.98 } gen_fixed bulbasaur "Bulbasaur, the seed Pokémon. It can be seen napping in bright sunlight. There is a seed on its back. By soaking up the sun's rays, the seed grows progressively larger." gen_fixed charizard "Charizard, the flame Pokémon. Charizard flies around the sky in search of powerful opponents. It breathes fire of such great heat that it melts anything." gen_fixed pikachu "Pikachu, the mouse Pokémon. When several of these Pokémon gather, their electricity could build and cause lightning storms." # ---- Stage 4: robot-voice DSP filter passes ---- # robot_filter.py -> heavy/original (square-wave ring mod + bitcrush + narrow bandpass) # robot_filter_light.py -> barely-there (sine ring mod, wide bandpass, mild drive) # robot_filter_v2.py -> parameterized 0..1 intensity knob (used at 0.45 = "medium") for name in bulbasaur charizard pikachu; do python3 robot_filter.py "${name}_raw.wav" "${name}_robot.wav" python3 robot_filter_light.py "${name}_neural.wav" "${name}_neural_light.wav" python3 robot_filter_v2.py "${name}_fixed.wav" "${name}_fixed_robot.wav" 0.45 done # ---- Stage 5: mp3 for easy playback/delivery ---- for f in *_raw.wav *_robot.wav *_neural.wav *_neural_light.wav *_fixed.wav *_fixed_robot.wav; do [ -f "$f" ] && ffmpeg -y -loglevel error -i "$f" -codec:a libmp3lame -qscale:a 4 "${f%.wav}.mp3" done echo "Done. See *.mp3 for output."