#!/usr/bin/env python3
"""
Modulo de agregacao avancada dos pageviews da landing (UTM/click-ids).

Le /opt/mia/workspace/clientes/px3lab/tracking/pageviews.jsonl e retorna:
- curva diaria
- distribuicao por utm_source
- top campanhas (utm_campaign)
- top paginas visitadas
- % com fbclid vs sem
"""
from __future__ import annotations

import json
from datetime import datetime, timedelta, timezone
from pathlib import Path
from typing import Iterable
from zoneinfo import ZoneInfo

TZ_BR = ZoneInfo("America/Sao_Paulo")
TZ_UTC = ZoneInfo("UTC")

PAGEVIEWS_FILE = Path("/opt/mia/workspace/clientes/px3lab/tracking/pageviews.jsonl")


def _iter_pageviews() -> Iterable[dict]:
    if not PAGEVIEWS_FILE.exists():
        return
    try:
        with PAGEVIEWS_FILE.open("r", encoding="utf-8") as f:
            for line in f:
                line = line.strip()
                if not line:
                    continue
                try:
                    yield json.loads(line)
                except Exception:
                    continue
    except Exception:
        return


def _parse_dt(v: str):
    if not v:
        return None
    try:
        return datetime.fromisoformat(v.replace("Z", "+00:00"))
    except Exception:
        return None


def _norm_page(p: str) -> str:
    p = (p or "").strip()
    if not p:
        return "/"
    # remove querystring residual e trailing slash duplicado
    p = p.split("?")[0]
    # Corta host se veio absoluto
    if p.startswith("http"):
        try:
            from urllib.parse import urlparse
            p = urlparse(p).path or "/"
        except Exception:
            pass
    return p or "/"


def agregar(inicio_utc: datetime, fim_utc: datetime) -> dict:
    """Le todos os pageviews e agrega no periodo."""
    total = 0
    com_fbclid = 0
    por_source: dict[str, int] = {}
    por_campanha: dict[str, int] = {}
    por_pagina: dict[str, int] = {}
    por_dia: dict[str, int] = {}

    for pv in _iter_pageviews():
        dt = _parse_dt(pv.get("ts", ""))
        if not dt or not (inicio_utc <= dt <= fim_utc):
            continue
        total += 1

        if pv.get("fbclid"):
            com_fbclid += 1

        src = (pv.get("utm_source") or "direto/organico").strip().lower() or "direto/organico"
        por_source[src] = por_source.get(src, 0) + 1

        camp = (pv.get("utm_campaign") or "").strip().lower()
        if camp:
            por_campanha[camp] = por_campanha.get(camp, 0) + 1

        pg = _norm_page(pv.get("page") or pv.get("path") or "/")
        por_pagina[pg] = por_pagina.get(pg, 0) + 1

        data_br = dt.astimezone(TZ_BR).strftime("%Y-%m-%d")
        por_dia[data_br] = por_dia.get(data_br, 0) + 1

    def _ranking(d: dict[str, int], key_name: str, top: int = 10) -> list[dict]:
        if not d:
            return []
        soma = sum(d.values()) or 1
        rows = [
            {key_name: k, "qtd": v, "pct": round(v / soma * 100, 1)}
            for k, v in d.items()
        ]
        rows.sort(key=lambda x: -x["qtd"])
        return rows[:top]

    curva = []
    for d in sorted(por_dia.keys()):
        try:
            label = datetime.strptime(d, "%Y-%m-%d").strftime("%d/%m")
        except Exception:
            label = d
        curva.append({"data": label, "data_iso": d, "pageviews": por_dia[d]})

    pct_fbclid = round(com_fbclid / total * 100, 1) if total else 0.0

    return {
        "total": total,
        "com_fbclid": com_fbclid,
        "sem_fbclid": total - com_fbclid,
        "pct_fbclid": pct_fbclid,
        "curva_diaria": curva,
        "por_source": _ranking(por_source, "source", top=10),
        "por_campanha": _ranking(por_campanha, "campanha", top=10),
        "por_pagina": _ranking(por_pagina, "pagina", top=15),
        "gerado_em": datetime.now(TZ_UTC).isoformat(timespec="seconds"),
    }
