import fitz
import json

doc = fitz.open(r'c:\fuentes\proyectos delphi\RAUL_ASENCIO\GESTION\documentacion raul asencio\guion_podcast_gustavo_ernesto_a5_un_bloque_por_pagina.pdf')

parsed_data = []

# Page 1: Cover
parsed_data.append({
    'page': 1,
    'type': 'cover',
    'title': 'CONVERSACIONES EN OBRADOR',
    'guests': 'Gustavo Mira y Ernesto Mira',
    'subtitle': 'Guión completo de preguntas',
    'format': 'Formato horizontal para mesa de podcast / grabación',
    'version': 'Versión de entrevista: un bloque temático por página, con pregunta principal destacada y repreguntas debajo.',
    'note': 'Preparado para lectura cómoda en mesa durante grabación'
})

# Page 2: Index
index_lines = [l.strip() for l in doc[1].get_text().split('\n') if l.strip()]
# lines:
# [0] Conversaciones en Obrador - Gustavo Mira y Ernesto Mira
# [1] 2
# [2] Mapa Del Guion
# [3..24] 1. ... to 22. Cierre
# [25] Consejo de uso
# [26..] Cada pagina...
items = []
advice = []
is_advice = False
for l in index_lines[3:]:
    if 'Consejo de uso' in l:
        is_advice = True
        continue
    if not is_advice:
        items.append(l)
    else:
        advice.append(l)

parsed_data.append({
    'page': 2,
    'type': 'index',
    'header': 'Conversaciones en Obrador - Gustavo Mira y Ernesto Mira',
    'title': 'Mapa Del Guión',
    'items': items,
    'advice': ' '.join(advice)
})

# Pages 3 to 24: Question blocks
for pno in range(2, 24):
    text = doc[pno].get_text()
    lines = [l.strip() for l in text.split('\n') if l.strip()]
    header = lines[0]
    page_num = int(lines[1])
    title = lines[2]
    
    # Process the remaining lines
    rest = lines[3:]
    main_lines = []
    repreguntas = []
    
    cur_rep_tag = None
    cur_rep_lines = []
    
    for l in rest:
        if l in ('REPREGUNTA / APOYO', 'ULTIMA'):
            if cur_rep_tag is not None:
                repreguntas.append({
                    'tag': cur_rep_tag,
                    'text': ' '.join(cur_rep_lines)
                })
                cur_rep_lines = []
            cur_rep_tag = l
        else:
            if cur_rep_tag is None:
                main_lines.append(l)
            else:
                cur_rep_lines.append(l)
                
    if cur_rep_tag is not None:
        repreguntas.append({
            'tag': cur_rep_tag,
            'text': ' '.join(cur_rep_lines)
        })
        
    main_text = ' '.join(main_lines)
    
    # Extract question number from main_text
    # Usually starts with "X. "
    main_num = ''
    main_body = main_text
    parts = main_text.split('.', 1)
    if len(parts) > 1 and parts[0].strip().isdigit():
        main_num = parts[0].strip()
        main_body = parts[1].strip()
        
    cleaned_repregs = []
    for r in repreguntas:
        rtxt = r['text']
        rnum = ''
        rbody = rtxt
        rparts = rtxt.split('.', 1)
        if len(rparts) > 1 and rparts[0].strip().isdigit():
            rnum = rparts[0].strip()
            rbody = rparts[1].strip()
        cleaned_repregs.append({
            'tag': r['tag'],
            'num': rnum,
            'body': rbody,
            'full_text': rtxt
        })
        
    parsed_data.append({
        'page': page_num,
        'type': 'block',
        'block_index': pno - 1, # 1..22
        'header': header,
        'title': title,
        'main_num': main_num,
        'main_body': main_body,
        'main_full': main_text,
        'repreguntas': cleaned_repregs
    })

# Page 25: Balas de repuesto
p25_lines = [l.strip() for l in doc[24].get_text().split('\n') if l.strip()]
p25_bullets = []
cur_bullet = []
p25_intro = []
for l in p25_lines[3:]:
    if l.startswith('-'):
        if cur_bullet:
            p25_bullets.append(' '.join(cur_bullet))
            cur_bullet = []
        cur_bullet.append(l[1:].strip())
    elif cur_bullet:
        cur_bullet.append(l)
    else:
        p25_intro.append(l)
if cur_bullet:
    p25_bullets.append(' '.join(cur_bullet))

parsed_data.append({
    'page': 25,
    'type': 'backup_bullets',
    'header': p25_lines[0],
    'title': p25_lines[2],
    'intro': ' '.join(p25_intro),
    'bullets': p25_bullets
})

with open('scratch/parsed_blocks.json', 'w', encoding='utf-8') as f:
    json.dump(parsed_data, f, ensure_ascii=False, indent=2)

print("Parsed successfully:", len(parsed_data), "pages.")
