from html.parser import HTMLParser
import xml.etree.ElementTree as ET
from xml.dom import minidom
from datetime import datetime
from zoneinfo import ZoneInfo
import re

INPUT = "/opt/epg/nova.html"
OUTPUT = "/opt/epg/sledovani.xml"

CHANNEL_ID = "nova"
CHANNEL_NAME = "Nova"

TIMEZONE = ZoneInfo("Europe/Prague")


# ==========================================
# HTML PARSER
# ==========================================

class EPGParser(HTMLParser):

    def __init__(self):
        super().__init__()

        self.programmes = []

        self.current_time = None
        self.current_title = None
        self.current_href = None

        self.in_li = False
        self.in_a = False

    def handle_starttag(self, tag, attrs):

        attrs = dict(attrs)

        # začiatok <li>
        if tag == "li":
            self.in_li = True
            self.current_time = None
            self.current_title = None
            self.current_href = None

        # <span data-time="...">
        if self.in_li and tag == "span":

            data_time = attrs.get("data-time")

            if data_time:
                try:
                    self.current_time = int(data_time)
                except ValueError:
                    pass

        # <a href="/content/detail/channelEvent:nova:...">
        if self.in_li and tag == "a":

            href = attrs.get("href", "")

            if "channelEvent:" + CHANNEL_ID + ":" in href:
                self.in_a = True
                self.current_href = href

    def handle_data(self, data):

        if self.in_a:
            text = data.strip()

            if text:
                if self.current_title:
                    self.current_title += " " + text
                else:
                    self.current_title = text

    def handle_endtag(self, tag):

        if tag == "a":
            self.in_a = False

        if tag == "li":

            if self.current_time and self.current_title:

                self.programmes.append({
                    "start": self.current_time,
                    "title": self.current_title,
                    "href": self.current_href
                })

            self.in_li = False


# ==========================================
# NAČÍTANIE HTML
# ==========================================

print()
print("Načítavam HTML:")
print(INPUT)

with open(INPUT, "r", encoding="utf-8") as f:
    html = f.read()


# ==========================================
# PARSOVANIE
# ==========================================

parser = EPGParser()
parser.feed(html)

programmes = parser.programmes


# odstránenie prípadných duplicít
unique = {}

for p in programmes:
    key = (p["start"], p["title"])

    if key not in unique:
        unique[key] = p

programmes = list(unique.values())

# zoradenie podľa času
programmes.sort(key=lambda x: x["start"])


print()
print("Nájdené programy:", len(programmes))


# ==========================================
# XMLTV
# ==========================================

tv = ET.Element("tv")
tv.set("generator-info-name", "SledovaniTV")


# CHANNEL
channel = ET.SubElement(tv, "channel")
channel.set("id", CHANNEL_ID)

display_name = ET.SubElement(channel, "display-name")
display_name.text = CHANNEL_NAME


# ==========================================
# PROGRAMY
# ==========================================

programmes_count = 0

for i, program in enumerate(programmes):

    start_timestamp = program["start"]

    # --------------------------------------
    # začiatok
    # --------------------------------------

    start_dt = datetime.fromtimestamp(
        start_timestamp,
        tz=TIMEZONE
    )

    # --------------------------------------
    # koniec = začiatok ďalšieho programu
    # --------------------------------------

    if i + 1 < len(programmes):

        end_timestamp = programmes[i + 1]["start"]

        end_dt = datetime.fromtimestamp(
            end_timestamp,
            tz=TIMEZONE
        )

    else:

        # posledný program - zatiaľ +1 hodina
        end_dt = start_dt.replace(
            minute=start_dt.minute
        )

        from datetime import timedelta
        end_dt = start_dt + timedelta(hours=1)


    start_xml = start_dt.strftime("%Y%m%d%H%M%S %z")
    end_xml = end_dt.strftime("%Y%m%d%H%M%S %z")


    # --------------------------------------
    # PROGRAM
    # --------------------------------------

    programme = ET.SubElement(tv, "programme")

    programme.set("start", start_xml)
    programme.set("stop", end_xml)
    programme.set("channel", CHANNEL_ID)


    title_element = ET.SubElement(
        programme,
        "title"
    )

    title_element.set("lang", "cs")
    title_element.text = program["title"]


    programmes_count += 1


# ==========================================
# ULOŽENIE XML
# ==========================================

xml_data = ET.tostring(
    tv,
    encoding="utf-8"
)

pretty = minidom.parseString(
    xml_data
).toprettyxml(
    indent="    ",
    encoding="UTF-8"
)


with open(OUTPUT, "wb") as f:
    f.write(pretty)


# ==========================================
# VÝPIS
# ==========================================

print()
print("====================================")
print(" SledovaniTV XMLTV")
print("====================================")
print(f"Kanál:    {CHANNEL_NAME}")
print(f"Programy: {programmes_count}")
print(f"Výstup:   {OUTPUT}")
print("====================================")
print()

# zobraz prvých 10 programov

for p in programmes[:10]:

    dt = datetime.fromtimestamp(
        p["start"],
        tz=TIMEZONE
    )

    print(
        dt.strftime("%d.%m.%Y %H:%M"),
        "-",
        p["title"]
    )

print()

