"""Create a separate narration draft with an explicit pronunciation dictionary.

Run: python prepare.py
Writes results.json and results.csv next to this script; never changes fixture.json.
No TTS model, shell command execution, API call, or language model is involved.
"""
import csv
import json
import re
from pathlib import Path


def prepare(text, pronunciations):
    if any(not key for key in pronunciations):
        raise ValueError('Pronunciation keys must not be empty')
    matches = []
    if pronunciations:
        # Longest first; literal matching; ASCII boundaries preserve identifiers
        # while allowing Korean particles such as API + 를.
        alternatives = '|'.join(re.escape(key) for key in sorted(pronunciations, key=len, reverse=True))
        pattern = re.compile(r'(?<![A-Za-z0-9_])(?:' + alternatives + r')(?![A-Za-z0-9_])')

        def replace(match):
            matches.append({'source': match.group(), 'spoken': pronunciations[match.group()]})
            return pronunciations[match.group()]

        spoken = pattern.sub(replace, text)
    else:
        spoken = text
    # A review queue, not an automatic pronunciation-quality verdict.
    unresolved = re.findall(r'[A-Za-z0-9_]+(?:[./:+-][A-Za-z0-9_]+)*', spoken)
    return {
        'original': text, 'spoken': spoken, 'replacements': matches,
        'unresolved': unresolved, 'needs_text_review': bool(unresolved),
    }


def main():
    root = Path(__file__).resolve().parent
    fixture = json.loads((root / 'fixture.json').read_text(encoding='utf-8'))
    results = []
    for case in fixture['cases']:
        result = {'id': case['id'], **prepare(case['original'], fixture['pronunciations'])}
        result['matches_expected_text'] = result['spoken'] == case['expected']
        result['matches_expected_review'] = result['needs_text_review'] == case['expected_review']
        results.append(result)
    (root / 'results.json').write_text(json.dumps(results, ensure_ascii=False, indent=2) + '\n', encoding='utf-8')
    with (root / 'results.csv').open('w', encoding='utf-8-sig', newline='') as stream:
        fields = ['id', 'original', 'spoken', 'needs_text_review', 'matches_expected_text', 'matches_expected_review']
        writer = csv.DictWriter(stream, fieldnames=fields, extrasaction='ignore')
        writer.writeheader()
        writer.writerows(results)
    print('Cases:', len(results))
    print('Expected text matches:', sum(r['matches_expected_text'] for r in results))
    print('Text review needed:', sum(r['needs_text_review'] for r in results))
    print('Audio generated: 0 (text preprocessing only)')


if __name__ == '__main__':
    main()
