import json, re
from pathlib import Path
base=Path('/home/hermes/workspace/clientes/chacara_sonho_verde/videos_legenda_drive_20261003')
d=json.loads((base/'transcript/source.json').read_text())
# Correct clear ASR orthographic/brand errors while preserving spoken content.
corrections={
 'chá cara':'Chácara', 'cara':'Chácara', 'top,':'Top,', 'top':'Top',
 'valinhos,':'Valinhos,', 'São':'São', 'Paulo.':'Paulo.',
 'polhas':'poliesportivas', 'portivas,':'portivas,', 'alenha,':'a lenha,',
 'Conheça':'Conheça', 'incrível?':'incrível?'
}
segs=[s for s in d['segments'] if s.get('words') and s['start']<52]
# Normalize words, maintaining token count/timestamps.
def norm(w):
    w=w.strip()
    if w.lower()=='chá': return 'Chácara'
    if w.lower()=='cara': return 'Chácara'
    if w.lower()=='top,': return 'Top,'
    if w.lower()=='top': return 'Top'
    if w.lower()=='polhas': return 'poliesportivas'
    if w.lower()=='portivas,': return 'portivas,'
    if w.lower()=='alenha,': return 'a lenha,'
    if w.lower()=='valinhos,': return 'Valinhos,'
    return w

def ass_time(t):
    h=int(t//3600); m=int((t%3600)//60); s=t%60
    return f'{h}:{m:02d}:{s:05.2f}'
header='''[Script Info]\nScriptType: v4.00+\nPlayResX: 606\nPlayResY: 1080\nScaledBorderAndShadow: yes\nWrapStyle: 2\n\n[V4+ Styles]\nFormat: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding\nStyle: Karaoke,DejaVu Sans,38,&H00FFFFFF,&H00FFFFFF,&H00101010,&H90101010,1,0,0,0,100,100,0,0,1,3,2,2,34,34,110,1\n\n[Events]\nFormat: Layer, Start, End, Style, Name, MarginL, MarginR, Effect, Text\n'''
lines=[]
for si,s in enumerate(segs):
    words=s['words']; n=len(words)
    # Keep each caption line within the narrow 606px portrait frame.
    sizes=[]; acc=0; chars=0
    for w in words:
        token=len(norm(w['word'])) + (1 if acc else 0)
        if acc and chars + token > 28:
            sizes.append(acc); acc=0; chars=0
        acc += 1; chars += token
    if acc: sizes.append(acc)
    pos=0
    for size in sizes:
        chunk=words[pos:pos+size]; pos+=size
        texts=[norm(w['word']) for w in chunk]
        full=' '.join(texts)
        # one event per active word; white inactive, yellow active
        for i,w in enumerate(chunk):
            start=float(w['start']); end=float(chunk[i+1]['start']) if i+1<len(chunk) else float(w['end'])
            if end<=start: end=start+0.08
            rendered=[]
            for j,t in enumerate(texts):
                color='&H0000FFFF' if j==i else '&H00FFFFFF' # ASS BGR: yellow
                rendered.append('{\\c'+color+'}'+t+'{\\c&H00FFFFFF}')
            lines.append(f'Dialogue: 0,{ass_time(start)},{ass_time(end)},Karaoke,,0,0,,{" ".join(rendered)}')
    # no further processing
(base/'karaoke.ass').write_text(header+'\n'.join(lines)+'\n',encoding='utf-8')
(base/'caption_text.txt').write_text('\n'.join(' '.join(norm(w['word']) for w in s['words']) for s in segs),encoding='utf-8')
print(f'wrote {len(lines)} timed word events from {sum(len(s["words"]) for s in segs)} words')
