#!/usr/bin/env python3
"""
INPI - Consulta de marca "Digital Mais" / variantes
Faz busca via Playwright na base pública busca.inpi.gov.br
Coleta dados brutos, salva CSV/JSON, tira screenshots.
"""

import json
import csv
import sys
import time
import re
import traceback
from pathlib import Path
from datetime import datetime
from playwright.sync_api import sync_playwright, TimeoutError as PWTimeout

BASE_DIR = Path("/opt/mia/workspace/temp")
SHOT_DIR = BASE_DIR / "inpi_screenshots"
SHOT_DIR.mkdir(parents=True, exist_ok=True)

CSV_OUT = BASE_DIR / "inpi_digital_mais_resultados.csv"
JSON_OUT = BASE_DIR / "inpi_digital_mais_resultados.json"
LOG_OUT = BASE_DIR / "inpi_digital_mais_run.log"

VARIACOES = [
    ("digital mais",      "radical"),   # radical amplo
    ("digital mais",      "exata"),     # exata
    ("digitalmais",       "exata"),
    ("digitalmais",       "radical"),
    ("mais digital",      "exata"),
    ("mais digital",      "radical"),
    ("digital more",      "exata"),
    ("digital more",      "radical"),
    ("digital +",         "exata"),
]

CLASSES = ["", "9", "35", "41", "42"]  # "" = todas

INPI_LOGIN_URL = "https://busca.inpi.gov.br/pePI/servlet/LoginController?action=login"
INPI_SEARCH_URL = "https://busca.inpi.gov.br/pePI/jsp/marcas/Pesquisa_classe_basica.jsp"

LOG_FH = open(LOG_OUT, "w", encoding="utf-8")


def log(msg):
    ts = datetime.now().strftime("%H:%M:%S")
    line = f"[{ts}] {msg}"
    print(line, flush=True)
    LOG_FH.write(line + "\n")
    LOG_FH.flush()


def slug(s):
    return re.sub(r"[^a-z0-9]+", "_", s.lower()).strip("_")


def parse_result_table(page, contexto):
    """
    Extrai as marcas listadas na página de resultados.
    A página do INPI retorna uma tabela com colunas:
    Nº do Processo | Marca | Classe | Situação | ...
    Retorna lista de dicts.
    """
    results = []
    try:
        # A tabela de resultados fica em <table> após o cabeçalho. Vamos varrer todas.
        rows = page.query_selector_all("table tr")
        headers = None
        for r in rows:
            cells = r.query_selector_all("td, th")
            if not cells:
                continue
            texts = [c.inner_text().strip() for c in cells]
            if not any(texts):
                continue
            # Detecta linha-cabeçalho
            joined = " | ".join(texts).lower()
            if headers is None and ("processo" in joined and "marca" in joined):
                headers = [t.lower() for t in texts]
                continue
            # linha de dados só se tiver um "processo" que parece número (>=6 dígitos)
            if headers and len(texts) >= 3:
                first = texts[0].replace(".", "").replace(" ", "")
                if re.match(r"^\d{6,}$", first):
                    row = {"raw_cells": texts}
                    for i, h in enumerate(headers):
                        if i < len(texts):
                            row[h] = texts[i]
                    row["_contexto"] = contexto
                    results.append(row)
    except Exception as e:
        log(f"  parse_result_table erro: {e}")
    return results


def count_indicator(page):
    """
    Tenta encontrar texto tipo 'X registros encontrados' ou similar.
    """
    try:
        body = page.content()
        m = re.search(r"(\d+)\s+registros?\s+encontrad", body, re.IGNORECASE)
        if m:
            return int(m.group(1))
        m = re.search(r"total\s+de\s+(\d+)", body, re.IGNORECASE)
        if m:
            return int(m.group(1))
    except Exception:
        pass
    return None


def do_search(page, marca, tipo, classe, screenshot_tag):
    """
    Executa 1 busca. tipo = 'exata' ou 'radical'.
    Retorna (resultados_lista, indicador_contagem, captcha_flag, erro_str)
    """
    ctx = f"{marca} / {tipo} / classe={classe or 'todas'}"
    log(f"  BUSCA: {ctx}")

    try:
        page.goto(INPI_SEARCH_URL, wait_until="domcontentloaded", timeout=60000)
    except PWTimeout:
        return [], None, False, "timeout ao abrir search page"

    # Verifica se voltou pra login (sessão expirada)
    if "LoginController" in page.url or "Login" in (page.title() or ""):
        log("  sessao expirou, refazendo login anonimo")
        page.goto(INPI_LOGIN_URL, wait_until="domcontentloaded", timeout=60000)
        page.goto(INPI_SEARCH_URL, wait_until="domcontentloaded", timeout=60000)

    # CAPTCHA?
    html = page.content().lower()
    if "captcha" in html or "recaptcha" in html:
        page.screenshot(path=str(SHOT_DIR / f"captcha_{screenshot_tag}.png"))
        return [], None, True, "captcha detectado"

    # Preenche marca
    try:
        page.fill('input[name="marca"]', marca)
    except Exception as e:
        return [], None, False, f"nao achou input marca: {e}"

    # Radical/exata
    try:
        if tipo == "radical":
            page.check('input[name="buscaExata"][value="nao"]')
        else:
            page.check('input[name="buscaExata"][value="sim"]')
    except Exception as e:
        log(f"  aviso radio buscaExata: {e}")

    # Classe
    if classe:
        try:
            page.fill('input[name="classeInter"]', classe)
        except Exception as e:
            log(f"  aviso classeInter: {e}")
    else:
        try:
            page.fill('input[name="classeInter"]', "")
        except Exception:
            pass

    # per page 100
    try:
        page.select_option('select[name="registerPerPage"]', "100")
    except Exception:
        pass

    # Submit
    try:
        with page.expect_navigation(wait_until="domcontentloaded", timeout=90000):
            page.click('input[type="submit"][name="botao"]')
    except PWTimeout:
        return [], None, False, "timeout no submit"
    except Exception as e:
        return [], None, False, f"erro submit: {e}"

    # Espera table
    try:
        page.wait_for_selector("table", timeout=30000)
    except PWTimeout:
        pass

    # detecta captcha pós-submit
    html2 = page.content().lower()
    if "captcha" in html2 or "recaptcha" in html2:
        page.screenshot(path=str(SHOT_DIR / f"captcha_post_{screenshot_tag}.png"))
        return [], None, True, "captcha pos-submit"

    # detecta "sem resultados"
    if "nenhum resultado" in html2 or "nao foram encontrados" in html2 or "não foram encontrados" in html2:
        page.screenshot(path=str(SHOT_DIR / f"empty_{screenshot_tag}.png"))
        return [], 0, False, None

    # Parse
    count = count_indicator(page)
    results = parse_result_table(page, ctx)

    # Screenshot da primeira página
    try:
        page.screenshot(path=str(SHOT_DIR / f"result_{screenshot_tag}.png"), full_page=True)
    except Exception:
        pass

    # Paginação: se houver "Próxima" / links de página, itera até 5 páginas por busca
    max_pages = 5
    pages_done = 1
    while pages_done < max_pages:
        next_link = None
        # Links de "Próxima" no INPI aparecem como "Próxima" ou como número de página
        for sel in ['a:has-text("Próxima")', 'a:has-text("Proxima")', 'a:has-text(">>")']:
            try:
                el = page.query_selector(sel)
                if el:
                    next_link = el
                    break
            except Exception:
                pass
        if not next_link:
            break
        try:
            with page.expect_navigation(wait_until="domcontentloaded", timeout=60000):
                next_link.click()
            page.wait_for_selector("table", timeout=15000)
            more = parse_result_table(page, ctx + f" p{pages_done+1}")
            if not more:
                break
            results.extend(more)
            pages_done += 1
        except Exception as e:
            log(f"  paginacao parou: {e}")
            break

    log(f"    -> {len(results)} linhas coletadas (indicador={count})")
    return results, count, False, None


def main():
    started = time.time()
    all_results = []
    captcha_hit = False
    errors = []

    with sync_playwright() as p:
        browser = p.chromium.launch(
            headless=True,
            args=["--no-sandbox", "--disable-blink-features=AutomationControlled"],
        )
        context = browser.new_context(
            user_agent=(
                "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 "
                "(KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36"
            ),
            viewport={"width": 1400, "height": 900},
            locale="pt-BR",
        )
        page = context.new_page()

        # login anônimo
        log("Abrindo sessao anonima INPI...")
        try:
            page.goto(INPI_LOGIN_URL, wait_until="domcontentloaded", timeout=60000)
        except PWTimeout:
            log("Timeout no login inicial - tentando de novo em 30s")
            time.sleep(30)
            page.goto(INPI_LOGIN_URL, wait_until="domcontentloaded", timeout=90000)

        page.screenshot(path=str(SHOT_DIR / "00_home.png"))

        for marca, tipo in VARIACOES:
            for classe in CLASSES:
                # Só faz sentido testar "digital +" em modo exata (radical exige 3+ chars alfa)
                if marca == "digital +" and tipo == "radical":
                    continue
                # limite de tempo global
                if time.time() - started > 40 * 60:
                    log("TEMPO LIMITE 40 MIN atingido - parando.")
                    break

                tag = f"{slug(marca)}_{tipo}_c{classe or 'all'}"
                res, cnt, cap, err = do_search(page, marca, tipo, classe, tag)
                if cap:
                    captcha_hit = True
                    errors.append(f"CAPTCHA em {marca}/{tipo}/{classe}")
                    log("!! CAPTCHA - abortando (Renato precisa ver)")
                    break
                if err:
                    errors.append(f"{marca}/{tipo}/{classe}: {err}")
                if res:
                    all_results.extend(res)
                # espera curta entre buscas pra nao apanhar do INPI
                time.sleep(3)
            if captcha_hit:
                break

        browser.close()

    # Deduplicar por processo+marca
    seen = set()
    dedup = []
    for r in all_results:
        key = (
            r.get("nº do processo") or r.get("processo") or r.get("raw_cells", [""])[0],
            r.get("marca") or (r.get("raw_cells") or [None, None])[1],
        )
        if key in seen:
            continue
        seen.add(key)
        dedup.append(r)

    # Salvar JSON
    with open(JSON_OUT, "w", encoding="utf-8") as f:
        json.dump({
            "gerado_em": datetime.now().isoformat(),
            "total_linhas_brutas": len(all_results),
            "total_dedup": len(dedup),
            "captcha": captcha_hit,
            "erros": errors,
            "resultados": dedup,
        }, f, ensure_ascii=False, indent=2)

    # Salvar CSV
    if dedup:
        all_keys = set()
        for r in dedup:
            all_keys.update(r.keys())
        cols = sorted(all_keys)
        with open(CSV_OUT, "w", newline="", encoding="utf-8") as f:
            w = csv.DictWriter(f, fieldnames=cols, extrasaction="ignore")
            w.writeheader()
            for r in dedup:
                w.writerow({k: (json.dumps(v, ensure_ascii=False) if isinstance(v, (list, dict)) else v) for k, v in r.items()})
    else:
        with open(CSV_OUT, "w", encoding="utf-8") as f:
            f.write("sem_resultados\n")

    log(f"FIM. Bruto={len(all_results)} dedup={len(dedup)} captcha={captcha_hit} erros={len(errors)}")
    LOG_FH.close()
    print(json.dumps({
        "captcha": captcha_hit,
        "total": len(dedup),
        "erros": errors[:10],
    }, ensure_ascii=False))


if __name__ == "__main__":
    try:
        main()
    except Exception as e:
        traceback.print_exc()
        LOG_FH.write("FATAL: " + str(e) + "\n" + traceback.format_exc())
        LOG_FH.close()
        sys.exit(1)
