Files
model-training/tts/generate_synthetic_voice.py
2026-07-19 23:52:25 +04:00

232 lines
7.8 KiB
Python

"""
maven tts dataset generator
- reads sentences from txt files (one per mood, one sentence per line)
- synthesizes audio via qwen3-tts 1.7B VoiceDesign
- saves wavs + metadata.csv for vits2 training
- resume-aware: skips already-synthesized samples
structure:
data/
neutral.txt
happy.txt
thinking.txt
confused.txt
tired.txt
run on your workstation with 16GB VRAM.
requires: pip install qwen-tts soundfile torch
"""
import csv
import json
import argparse
import soundfile as sf
from pathlib import Path
from tqdm import tqdm
import torch
from ruaccent import RUAccent
from qwen_tts import Qwen3TTSModel
# ── config ────────────────────────────────────────────────────────────────────
MODEL = "Qwen/Qwen3-TTS-12Hz-1.7B-Base"
DATA_DIR = Path("tts/data") # where mood txt files live
OUT_DIR = Path("tts/dataset") # wav output
REF_NUM = "2"
REF_DIR = "tts/ref/" + REF_NUM
META_FILE = OUT_DIR / "metadata.csv"
CACHE_FILE = OUT_DIR / ".progress.json"
REF_WAV = REF_DIR + "/maven_reference.wav"
REF_TXT = REF_DIR + "/maven_reference.txt"
VOWELS = set('аеёиоуыэюяАЕЁИОУЫЭЮЯ')
VOICE_DESIGN = (
"Female voice, mid-20s. Natural and grounded, but lively. "
"Fast, slightly uneven pacing — thoughts often move ahead of speech. "
"Light, reactive expressiveness: small spikes of interest, brief amused or curious tones, "
"then returning to a calm baseline. "
"Occasional micro-pauses, restarts, or cut-off phrases, like thinking in real time. "
"Tone is clear and fairly light, but not high-pitched or cutesy. "
"Not performative — feels spontaneous rather than acted. "
"Subtle unpredictability in rhythm and delivery. "
"No anime-style, no exaggerated эмоции, no theatrical voice acting."
)
# mood → (emotion_id, qwen3-tts instruct)
MOODS = {
"neutral": (0, "Speak in a calm, matter-of-fact tone."),
"happy": (1, "Speak with genuine excitement, fast tempo, slightly higher pitch."),
"thinking": (2, "Speak slowly, thoughtfully, trailing off mid-sentence."),
"confused": (3, "Speak with puzzled, slightly impatient intonation."),
"tired": (4, "Speak softly, deflated, low energy."),
}
_accentor = RUAccent()
# ── helpers ───────────────────────────────────────────────────────────────────
def plus_to_acute(text: str) -> str:
result = []
i = 0
while i < len(text):
if text[i] == '+' and i + 1 < len(text) and text[i+1] in VOWELS:
result.append(text[i+1].upper())
i += 2
else:
result.append(text[i])
i += 1
return ''.join(result)
def preprocess(text: str) -> str:
result = _accentor.process_all(text)
return plus_to_acute(result)
def load_sentences(mood: str) -> list[str]:
path = DATA_DIR / f"{mood}-voice-dataset-list.txt"
if not path.exists():
print(f" [!] {path} not found, skipping")
return []
# lines = [preprocess(l.strip()) for l in path.read_text(encoding="utf-8").splitlines() if l.strip()]
lines = [l.strip() for l in path.read_text(encoding="utf-8").splitlines() if l.strip()]
print(f" [*] {mood}: {len(lines)} sentences loaded from {path}")
return lines
def load_progress() -> dict:
if CACHE_FILE.exists():
return json.loads(CACHE_FILE.read_text())
return {}
def save_progress(progress: dict):
CACHE_FILE.write_text(json.dumps(progress, indent=2))
def synthesize(model: Qwen3TTSModel, text: str, instruct: str) -> tuple:
# full_instruct = f"{VOICE_DESIGN} {instruct}"
ref_text = Path(REF_TXT).read_text(encoding="utf-8").strip()
print(f"[*] synthesizing text: {text}")
wavs, sr = model.generate_voice_clone(
text=text,
ref_audio=REF_WAV,
ref_text=ref_text,
language="Russian",
x_vector_only_mode=True,
)
audio = wavs[0] if isinstance(wavs, list) else wavs
return audio, sr
# ── main ──────────────────────────────────────────────────────────────────────
def main():
parser = argparse.ArgumentParser(description="Generate maven TTS dataset from txt files")
parser.add_argument("--moods", nargs="+", default=list(MOODS.keys()),
help="Which moods to process (default: all)")
parser.add_argument("--limit", type=int, default=None,
help="Max sentences per mood (default: all)")
parser.add_argument("--dry-run", action="store_true",
help="Print sentences without synthesizing")
args = parser.parse_args()
OUT_DIR.mkdir(exist_ok=True)
for mood in MOODS:
(OUT_DIR / mood).mkdir(exist_ok=True)
progress = load_progress()
if not args.dry_run:
print(f"[*] loading tts model: {MODEL}")
model = Qwen3TTSModel.from_pretrained(
MODEL,
device_map="cuda:0",
dtype=torch.float32,
)
else:
model = None
print("[*] dry run — no synthesis")
# open metadata in append mode
meta_exists = META_FILE.exists()
meta_f = open(META_FILE, "a", newline="", encoding="utf-8")
writer = csv.DictWriter(meta_f, fieldnames=["file_path", "text", "mood", "emotion_id"])
if not meta_exists:
writer.writeheader()
total_written = 0
try:
for mood in args.moods:
if mood not in MOODS:
print(f"[!] unknown mood '{mood}', skipping")
continue
emotion_id, instruct = MOODS[mood]
sentences = load_sentences(mood)
if not sentences:
continue
if args.limit:
sentences = sentences[:args.limit]
done = set(progress.get(mood, []))
todo = [s for s in sentences if s not in done]
if not todo:
print(f"[=] {mood}: all {len(sentences)} already done")
continue
print(f"\n[>] {mood}: {len(todo)} to synthesize ({len(done)} already done)")
for actual_idx, text in enumerate(tqdm(todo, desc=mood)):
if text in done:
continue
if args.limit and (actual_idx - len(done)) >= args.limit:
break
out_path = OUT_DIR / mood / f"{actual_idx:04d}.wav"
if out_path.exists():
done.add(text)
continue
if args.dry_run:
print(f" [{mood}/{actual_idx:04d}] {text}")
continue
try:
audio, sr = synthesize(model, text, instruct)
sf.write(str(out_path), audio, sr)
writer.writerow({
"file_path": str(out_path),
"text": text,
"mood": mood,
"emotion_id": emotion_id,
})
meta_f.flush()
done.add(text)
total_written += 1
except Exception as e:
print(f"\n [!] failed on '{text[:40]}...': {e}")
continue
progress[mood] = list(done)
save_progress(progress)
print(f" [✓] {mood} done")
finally:
meta_f.close()
if not args.dry_run:
print(f"\n[✓] wrote {total_written} new samples → {OUT_DIR}/")
print(f" metadata → {META_FILE}")
if __name__ == "__main__":
main()