reddex/saved/pokedex_voice_spike_scripts/generate_samples.sh
forgejoadmin 71db2d1ab9 Initial commit
Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-08-18 03:16:42 -04:00

74 lines
4.3 KiB
Bash

#!/usr/bin/env bash
# Reproduces the Pokedex voice spike end to end:
# TTS (espeak-ng and Piper) -> robot-voice DSP filter -> mp3
#
# Setup (Debian/Ubuntu):
# sudo apt-get install -y espeak-ng ffmpeg
# pip install piper-tts --break-system-packages
# pip install numpy scipy --break-system-packages
#
# Voice model:
# Official route: download a Piper voice (.onnx + .onnx.json) from the
# Piper voices repo and point --model at it, e.g. en_US-lessac-medium.
# In this sandbox, huggingface.co wasn't reachable, so I instead pulled a
# community PyPI wheel that bundles the model file directly:
# pip download --no-deps joe-us-piper-voice
# (unzip the wheel; the .onnx/.onnx.json live under
# joe_us_piper_voice/data/). Not an official source -- for production,
# get voices from Piper's own releases instead.
set -euo pipefail
cd "$(dirname "$0")"
VOICE_MODEL="voices/en_US-joe-medium.onnx"
export ALSA_CONFIG_PATH=/dev/null # silence ALSA warnings in headless envs
# ---- Stage 1: raw espeak-ng (formant synth, offline, inherently "robotic") ----
espeak-ng -v en-us -s 150 -w bulbasaur_raw.wav \
"Bulbasaur. Seed pokemon. It can be seen napping in bright sunlight. There is a seed on its back. By soaking up the sun's rays, the seed grows progressively larger."
espeak-ng -v en-us -s 150 -w charizard_raw.wav \
"Charizard. Flame pokemon. Charizard flies around the sky in search of powerful opponents. It breathes fire of such great heat that it melts anything."
espeak-ng -v en-us -s 150 -w pikachu_raw.wav \
"Pikachu. Mouse pokemon. When several of these pokemon gather, their electricity could build and cause lightning storms."
# ---- Stage 2: Piper neural TTS, first pass (natural but bad cadence on rare words) ----
gen_neural() {
local name="$1" text="$2"
echo "$text" | piper -m "$VOICE_MODEL" -f "${name}_neural.wav"
}
gen_neural bulbasaur "Bulbasaur. Seed pokemon. It can be seen napping in bright sunlight. There is a seed on its back. By soaking up the sun's rays, the seed grows progressively larger."
gen_neural charizard "Charizard. Flame pokemon. Charizard flies around the sky in search of powerful opponents. It breathes fire of such great heat that it melts anything."
gen_neural pikachu "Pikachu. Mouse pokemon. When several of these pokemon gather, their electricity could build and cause lightning storms."
# ---- Stage 3: Piper neural TTS, cadence-fixed pass ----
# Fix = (a) standard "X, the Y Pokemon." phrasing instead of two short
# sentences, which stopped "Pokemon" from landing phrase-final where
# duration models over-lengthen it, and (b) reduced noise-w-scale (duration
# randomness) and noise-scale (audio variance) so rare proper nouns don't
# get a random stretched/warped rendering.
gen_fixed() {
local name="$1" text="$2"
echo "$text" | piper -m "$VOICE_MODEL" -f "${name}_fixed.wav" \
--noise-scale 0.5 --noise-w-scale 0.3 --length-scale 0.98
}
gen_fixed bulbasaur "Bulbasaur, the seed Pokémon. It can be seen napping in bright sunlight. There is a seed on its back. By soaking up the sun's rays, the seed grows progressively larger."
gen_fixed charizard "Charizard, the flame Pokémon. Charizard flies around the sky in search of powerful opponents. It breathes fire of such great heat that it melts anything."
gen_fixed pikachu "Pikachu, the mouse Pokémon. When several of these Pokémon gather, their electricity could build and cause lightning storms."
# ---- Stage 4: robot-voice DSP filter passes ----
# robot_filter.py -> heavy/original (square-wave ring mod + bitcrush + narrow bandpass)
# robot_filter_light.py -> barely-there (sine ring mod, wide bandpass, mild drive)
# robot_filter_v2.py -> parameterized 0..1 intensity knob (used at 0.45 = "medium")
for name in bulbasaur charizard pikachu; do
python3 robot_filter.py "${name}_raw.wav" "${name}_robot.wav"
python3 robot_filter_light.py "${name}_neural.wav" "${name}_neural_light.wav"
python3 robot_filter_v2.py "${name}_fixed.wav" "${name}_fixed_robot.wav" 0.45
done
# ---- Stage 5: mp3 for easy playback/delivery ----
for f in *_raw.wav *_robot.wav *_neural.wav *_neural_light.wav *_fixed.wav *_fixed_robot.wav; do
[ -f "$f" ] && ffmpeg -y -loglevel error -i "$f" -codec:a libmp3lame -qscale:a 4 "${f%.wav}.mp3"
done
echo "Done. See *.mp3 for output."