import os
import json
import logging
import re
import threading
import time
import uuid
import re, unicodedata
from pathlib import Path
from datetime import datetime, timezone
from http.server import ThreadingHTTPServer, SimpleHTTPRequestHandler

import httpx
from telegram import Update
from telegram.ext import Application, MessageHandler, ContextTypes, filters

from feedgen.feed import FeedGenerator

try:
    from mutagen import File as MutagenFile
except ImportError:
    MutagenFile = None

logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
log = logging.getLogger("podcast-bot")

DATA_DIR = Path(os.environ.get("DATA_DIR", "/data"))
EPISODES_DIR = DATA_DIR / "episodes"
EPISODES_JSON = DATA_DIR / "episodes.json"
FEED_PATH = DATA_DIR / "feed.xml"

BOT_TOKEN = os.environ["TELEGRAM_BOT_TOKEN"]
ALLOWED_CHAT_ID = os.environ.get("ALLOWED_CHAT_ID", "").strip()

LOCAL_API_BASE = os.environ.get("LOCAL_API_BASE", "").strip().rstrip("/")

# Carpeta donde el add-on del servidor local de Telegram deja los archivos,
# montada dentro de este contenedor (ver docker run -v ... :/tgfiles:ro).
TG_FILES_DIR = os.environ.get("TG_FILES_DIR", "/tgfiles")

PUBLIC_BASE_URL = os.environ["PUBLIC_BASE_URL"].rstrip("/")
HTTP_PORT = int(os.environ.get("HTTP_PORT", "8095"))

PODCAST_TITLE = os.environ.get("PODCAST_TITLE", "Mi Podcast")
PODCAST_DESC = os.environ.get("PODCAST_DESCRIPTION", "Generado automaticamente desde Telegram")
PODCAST_AUTHOR = os.environ.get("PODCAST_AUTHOR", "Autor")
PODCAST_LANGUAGE = os.environ.get("PODCAST_LANGUAGE", "es")
PODCAST_COVER = os.environ.get("PODCAST_COVER_URL", f"{PUBLIC_BASE_URL}/cover.jpg")
PODCAST_CATEGORY = os.environ.get("PODCAST_CATEGORY", "Society & Culture")
PODCAST_EXPLICIT = os.environ.get("PODCAST_EXPLICIT", "no")

EPISODES_DIR.mkdir(parents=True, exist_ok=True)


class PodcastRequestHandler(SimpleHTTPRequestHandler):
    def __init__(self, *args, **kwargs):
        super().__init__(*args, directory=str(DATA_DIR), **kwargs)

    def guess_type(self, path):
        if str(path).endswith(".xml"):
            return "application/rss+xml"
        return super().guess_type(path)

    def log_message(self, fmt, *args):
        log.info("HTTP " + fmt, *args)


def start_web_server():
    server = ThreadingHTTPServer(("0.0.0.0", HTTP_PORT), PodcastRequestHandler)
    log.info("Servidor web escuchando en el puerto %s", HTTP_PORT)
    server.serve_forever()


def load_episodes():
    if EPISODES_JSON.exists():
        return json.loads(EPISODES_JSON.read_text(encoding="utf-8"))
    return []


def save_episodes(episodes):
    EPISODES_JSON.write_text(
        json.dumps(episodes, indent=2, ensure_ascii=False), encoding="utf-8"
    )


def get_duration_seconds(filepath: Path):
    if MutagenFile is None:
        return None
    try:
        audio = MutagenFile(str(filepath))
        if audio is not None and audio.info is not None:
            return int(audio.info.length)
    except Exception as e:
        log.warning("No se pudo leer duracion de %s: %s", filepath, e)
    return None


def format_duration(seconds):
    if seconds is None:
        return "00:00:00"
    h = seconds // 3600
    m = (seconds % 3600) // 60
    s = seconds % 60
    return f"{h:02d}:{m:02d}:{s:02d}"


def rebuild_feed():
    episodes = load_episodes()
    fg = FeedGenerator()
    fg.load_extension("podcast")
    fg.title(PODCAST_TITLE)
    fg.link(href=PUBLIC_BASE_URL, rel="alternate")
    fg.description(PODCAST_DESC)
    fg.language(PODCAST_LANGUAGE)
    fg.podcast.itunes_author(PODCAST_AUTHOR)
    fg.podcast.itunes_category(PODCAST_CATEGORY)
    fg.podcast.itunes_explicit(PODCAST_EXPLICIT)
    fg.podcast.itunes_image(PODCAST_COVER)
    fg.image(url=PODCAST_COVER, title=PODCAST_TITLE, link=PUBLIC_BASE_URL)

    for ep in sorted(episodes, key=lambda e: e["pub_date"], reverse=True):
        fe = fg.add_entry()
        fe.id(ep["guid"])
        fe.title(ep["title"])
        fe.description(ep.get("description", ep["title"]))
        fe.enclosure(ep["url"], str(ep["filesize"]), ep["mime_type"])
        fe.pubDate(ep["pub_date"])
        fe.podcast.itunes_duration(ep.get("duration_str", "00:00:00"))

    fg.rss_file(str(FEED_PATH))
    log.info("Feed regenerado con %d episodios", len(episodes))


async def download_with_progress(url: str, dest_path: Path, status_msg, size_hint=None):
    last_percent = -1
    last_edit_ts = 0.0
    downloaded = 0
    async with httpx.AsyncClient(timeout=None, follow_redirects=True) as client:
        async with client.stream("GET", url) as resp:
            resp.raise_for_status()
            total = int(resp.headers.get("Content-Length") or 0) or (size_hint or 0)
            with open(dest_path, "wb") as f:
                async for chunk in resp.aiter_bytes(chunk_size=256 * 1024):
                    f.write(chunk)
                    downloaded += len(chunk)
                    if total:
                        percent = min(99, int(downloaded * 100 / total))
                        now = time.monotonic()
                        if percent >= last_percent + 5 and now - last_edit_ts >= 2:
                            last_percent = percent
                            last_edit_ts = now
                            filled = percent // 10
                            bar = "█" * filled + "░" * (10 - filled)
                            try:
                                await status_msg.edit_text(f"Descargando... {percent}%\n[{bar}]")
                            except Exception:
                                pass
    return downloaded


async def copy_with_progress(src: Path, dest_path: Path, status_msg):
    total = src.stat().st_size
    copied = 0
    last_percent = -1
    last_edit_ts = 0.0
    with open(src, "rb") as fin, open(dest_path, "wb") as fout:
        while True:
            chunk = fin.read(1024 * 1024)
            if not chunk:
                break
            fout.write(chunk)
            copied += len(chunk)
            if total:
                percent = min(99, int(copied * 100 / total))
                now = time.monotonic()
                if percent >= last_percent + 10 and now - last_edit_ts >= 2:
                    last_percent = percent
                    last_edit_ts = now
                    filled = percent // 10
                    bar = "█" * filled + "░" * (10 - filled)
                    try:
                        await status_msg.edit_text(f"Guardando... {percent}%\n[{bar}]")
                    except Exception:
                        pass
    return copied


DATE_PATTERNS = [
    (re.compile(r"(\d{1,2})[/-](\d{1,2})[/-](\d{4})"), ("d", "m", "y")),   # 15/03/2026
    (re.compile(r"(\d{4})[/-](\d{1,2})[/-](\d{1,2})"), ("y", "m", "d")),   # 2026-03-15
]


def extract_date(caption: str):
    """Busca una fecha en el pie del audio.

    Devuelve (fecha, texto_sin_la_fecha). Si no hay fecha, (None, caption).
    Acepta 15/03/2026, 15-03-2026, 2026-03-15 y opcionalmente hora 21:30.
    """
    if not caption:
        return None, caption

    for pattern, order in DATE_PATTERNS:
        m = pattern.search(caption)
        if not m:
            continue
        parts = dict(zip(order, m.groups()))
        try:
            day = int(parts["d"])
            month = int(parts["m"])
            year = int(parts["y"])
        except (KeyError, ValueError):
            continue

        rest = caption[: m.start()] + caption[m.end():]

        hour, minute = 12, 0
        tm = re.search(r"\b(\d{1,2}):(\d{2})\b", rest)
        if tm:
            try:
                hour = int(tm.group(1))
                minute = int(tm.group(2))
                rest = rest[: tm.start()] + rest[tm.end():]
            except ValueError:
                pass

        try:
            dt = datetime(year, month, day, hour, minute, tzinfo=timezone.utc)
        except ValueError:
            return None, caption

        rest = rest.replace("|", " ").replace("-", " ") if rest.strip(" |-") == "" else rest
        rest = re.sub(r"\s*\|\s*$", "", rest.strip())
        rest = re.sub(r"^\s*\|\s*", "", rest).strip(" -–|")
        return dt, rest

    return None, caption


def find_local_source(raw_path: str):
    """Si el servidor local ya dejo el archivo en disco, devuelve su ruta."""
    marker = "/telegram-bot-api/"
    if marker not in raw_path:
        return None
    rest = raw_path.split(marker, 1)[1]
    candidate = Path(TG_FILES_DIR) / rest
    return candidate if candidate.exists() else None


async def handle_audio(update: Update, context: ContextTypes.DEFAULT_TYPE):
    chat_id = update.effective_chat.id
    if ALLOWED_CHAT_ID and str(chat_id) != ALLOWED_CHAT_ID:
        log.info("Mensaje ignorado de chat no autorizado: %s", chat_id)
        return

    msg = update.effective_message
    tg_file = None
    file_name = None
    mime_type = "audio/mpeg"

    if msg.audio:
        tg_file = await msg.audio.get_file()
        file_name = msg.audio.file_name or f"{msg.audio.file_unique_id}.mp3"
        mime_type = msg.audio.mime_type or mime_type
    elif msg.voice:
        tg_file = await msg.voice.get_file()
        file_name = f"{msg.voice.file_unique_id}.ogg"
        mime_type = msg.voice.mime_type or "audio/ogg"
    elif msg.document and (msg.document.mime_type or "").startswith("audio"):
        tg_file = await msg.document.get_file()
        file_name = msg.document.file_name or f"{msg.document.file_unique_id}"
        mime_type = msg.document.mime_type or mime_type
    else:
        return

    status_msg = await msg.reply_text("Descargando... 0%")

    ep_id = uuid.uuid4().hex[:12]
    ext = Path(file_name).suffix or ".mp3"
    base = unicodedata.normalize("NFKD", Path(file_name).stem).encode("ascii", "ignore").decode()
    base = re.sub(r"[^A-Za-z0-9._-]+", "_", base).strip("._-")[:80] or ep_id
    stored_name = f"{base}{ext}"
    n = 2
    while (EPISODES_DIR / stored_name).exists():
        stored_name = f"{base}_{n}{ext}"
        n += 1
    dest_path = EPISODES_DIR / stored_name

    raw_path = tg_file.file_path or ""
    local_src = find_local_source(raw_path)

    try:
        if local_src is not None:
            log.info("Copiando desde el servidor local: %s", local_src)
            await copy_with_progress(local_src, dest_path, status_msg)
        else:
            if raw_path.startswith("http"):
                file_url = raw_path
            else:
                file_url = f"https://api.telegram.org/file/bot{BOT_TOKEN}/{raw_path}"
            log.info("Descargando por HTTP: %s", file_url)
            await download_with_progress(
                file_url, dest_path, status_msg,
                size_hint=getattr(tg_file, "file_size", None),
            )
    except Exception as e:
        log.exception("Fallo al obtener el audio")
        await status_msg.edit_text(f"No se pudo guardar el audio: {e}")
        return

    custom_date, caption_rest = extract_date(msg.caption or "")
    title = (caption_rest or Path(file_name).stem or f"Episodio {ep_id}").strip()
    duration_seconds = get_duration_seconds(dest_path)
    filesize = dest_path.stat().st_size
    pub_date = custom_date or datetime.now(timezone.utc)

    episodes = load_episodes()
    episodes.append(
        {
            "guid": ep_id,
            "title": title,
            "description": title,
            "url": f"{PUBLIC_BASE_URL}/episodes/{stored_name}",
            "filesize": filesize,
            "mime_type": mime_type,
            "pub_date": pub_date.isoformat(),
            "duration_str": format_duration(duration_seconds),
        }
    )
    save_episodes(episodes)
    rebuild_feed()

    fecha_txt = pub_date.strftime("%d/%m/%Y")
    await status_msg.edit_text(
        f"Episodio publicado: {title}\nFecha: {fecha_txt}\nFeed: {PUBLIC_BASE_URL}/feed.xml"
    )


def main():
    if not EPISODES_JSON.exists():
        save_episodes([])
    rebuild_feed()

    threading.Thread(target=start_web_server, daemon=True).start()

    builder = (
        Application.builder()
        .token(BOT_TOKEN)
        .connect_timeout(60)
        .read_timeout(600)
        .write_timeout(600)
        .pool_timeout(60)
    )
    if LOCAL_API_BASE:
        builder = builder.base_url(f"{LOCAL_API_BASE}/bot").base_file_url(
            f"{LOCAL_API_BASE}/file/bot"
        )
        log.info("Usando servidor local de Telegram Bot API en %s", LOCAL_API_BASE)

    app = builder.build()
    app.add_handler(
        MessageHandler(
            filters.AUDIO | filters.VOICE | filters.Document.AUDIO,
            handle_audio,
        )
    )
    log.info("Bot iniciado. Esperando audios...")
    app.run_polling()


if __name__ == "__main__":
    main()
