"""Synthesize each lab sentence twice (original vs spoken) with Supertonic 3, then STT both.

Run next to results.json from prepare.py:
    pip install supertonic==1.3.1 mlx-whisper   # mlx-whisper is Apple Silicon only
    python synthesize_ab.py
Writes audio-ab/T*_original.wav, T*_spoken.wav and ab_results.csv.
Same model, voice, steps and speed for both inputs so only the text differs.
"""
import csv
import json
import time
from pathlib import Path

from supertonic import TTS

root = Path(__file__).resolve().parent
rows = json.loads((root.parent / 'results.json').read_text(encoding='utf-8'))
tts = TTS(model='supertonic-3')
voice = tts.get_voice_style('F1')
out = []
for row in rows:
    for kind in ('original', 'spoken'):
        start = time.time()
        wav, duration = tts.synthesize(row[kind], voice_style=voice, lang='ko')  # total_steps=8, speed=1.05 (defaults)
        path = root / f"{row['id']}_{kind}.wav"
        tts.save_audio(wav, str(path))
        out.append({'id': row['id'], 'kind': kind, 'text': row[kind],
                    'generate_s': round(time.time() - start, 3), 'audio_s': round(float(duration[0]), 3)})

try:
    import mlx_whisper
    for item in out:
        result = mlx_whisper.transcribe(str(root / f"{item['id']}_{item['kind']}.wav"),
                                        path_or_hf_repo='mlx-community/whisper-large-v3-turbo',
                                        language='ko', temperature=0.0)
        item['stt'] = result['text'].strip()
except ImportError:
    pass

with (root / 'ab_results.csv').open('w', encoding='utf-8-sig', newline='') as stream:
    writer = csv.DictWriter(stream, fieldnames=list(out[0]))
    writer.writeheader()
    writer.writerows(out)
