init
This commit is contained in:
@@ -0,0 +1,89 @@
|
||||
"""
|
||||
build the piper LJSpeech dataset from the teacher-generated dataset.
|
||||
|
||||
dataset/metadata.csv (file_path,text,mood,emotion_id) + 24kHz wavs
|
||||
→ piper/dataset/wav/<id>.wav (22050 Hz mono)
|
||||
→ piper/dataset/metadata.csv (id|stressed_text)
|
||||
|
||||
stress convention: the mood lists / teacher text use "+VOWEL" (plus before the
|
||||
stressed vowel). espeak-ng — piper's phonemizer — wants the COMBINING ACUTE
|
||||
(U+0301) AFTER the vowel. we convert here. (the uppercase-vowel form in
|
||||
generate_synthetic_voice.py's plus_to_acute is for the qwen/OmniVoice teacher,
|
||||
NOT espeak — don't reuse it.)
|
||||
|
||||
run: python piper/build_dataset.py # build
|
||||
python piper/build_dataset.py --check # self-check only
|
||||
requires: pip install soundfile librosa
|
||||
"""
|
||||
|
||||
import csv
|
||||
import sys
|
||||
import argparse
|
||||
from pathlib import Path
|
||||
|
||||
ACUTE = "́" # combining acute accent
|
||||
VOWELS = set("аеёиоуыэюяАЕЁИОУЫЭЮЯ")
|
||||
SRC_META = Path("tts/dataset/metadata.csv")
|
||||
OUT_DIR = Path("tts/piper/dataset")
|
||||
TARGET_SR = 22050
|
||||
|
||||
|
||||
def plus_to_espeak(text: str) -> str:
|
||||
"""'+VOWEL' -> 'VOWEL' + U+0301 (espeak-ng stress). Leaves other '+' intact."""
|
||||
out = []
|
||||
i = 0
|
||||
while i < len(text):
|
||||
if text[i] == "+" and i + 1 < len(text) and text[i + 1] in VOWELS:
|
||||
out.append(text[i + 1])
|
||||
out.append(ACUTE)
|
||||
i += 2
|
||||
else:
|
||||
out.append(text[i])
|
||||
i += 1
|
||||
return "".join(out)
|
||||
|
||||
|
||||
def build():
|
||||
import soundfile as sf
|
||||
import librosa
|
||||
|
||||
wav_out = OUT_DIR / "wav"
|
||||
wav_out.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
rows = list(csv.DictReader(SRC_META.open(encoding="utf-8")))
|
||||
meta_lines, n = [], 0
|
||||
for r in rows:
|
||||
src = Path(r["file_path"])
|
||||
if not src.exists():
|
||||
print(f" [!] missing wav, skipping: {src}")
|
||||
continue
|
||||
# id: <mood>-<stem> so ids stay unique across per-mood NNNN.wav collisions
|
||||
wid = f"{r['mood']}-{src.stem}"
|
||||
audio, sr = sf.read(str(src))
|
||||
if audio.ndim > 1:
|
||||
audio = audio.mean(axis=1)
|
||||
if sr != TARGET_SR:
|
||||
audio = librosa.resample(audio, orig_sr=sr, target_sr=TARGET_SR)
|
||||
sf.write(str(wav_out / f"{wid}.wav"), audio, TARGET_SR)
|
||||
meta_lines.append(f"{wid}|{plus_to_espeak(r['text'])}")
|
||||
n += 1
|
||||
|
||||
(OUT_DIR / "metadata.csv").write_text("\n".join(meta_lines) + "\n", encoding="utf-8")
|
||||
print(f"[✓] {n} samples → {OUT_DIR} (wav {TARGET_SR}Hz, metadata id|text)")
|
||||
|
||||
|
||||
def check():
|
||||
assert plus_to_espeak("з+амок") == "за" + ACUTE + "мок"
|
||||
assert plus_to_espeak("сто+ят") == "стоя" + ACUTE + "т"
|
||||
assert plus_to_espeak("нет плюса") == "нет плюса" # no stress → untouched
|
||||
assert plus_to_espeak("1 + 2") == "1 + 2" # '+' not before vowel → kept
|
||||
print("[✓] plus_to_espeak self-check passed")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--check", action="store_true", help="run self-check only")
|
||||
args = ap.parse_args()
|
||||
check()
|
||||
if not args.check:
|
||||
build()
|
||||
Reference in New Issue
Block a user