import gzip
import re
import urllib.request
import xml.etree.ElementTree as ET
import unicodedata

EPG_URL = "https://www.open-epg.com/files/poland1.xml.gz"
INPUT_M3U = "/opt/epg/POLSKA_TV.m3u"
OUTPUT_M3U = "/opt/epg/pl_epg.m3u"

print("Downloading EPG...")

data = urllib.request.urlopen(EPG_URL).read()
xml = gzip.decompress(data)

root = ET.fromstring(xml)

channels = {}

# ==========================================
# NORMALIZÁCIA NÁZVU
# ==========================================

def normalize(name):

    if not name:
        return ""

    name = name.strip().upper()

    # odstránenie diakritiky
    name = unicodedata.normalize("NFKD", name)
    name = "".join(
        c for c in name
        if not unicodedata.combining(c)
    )

    # odstránenie prefixov
    prefixes = [
        "PL|",
        "PL |",
        "POL|",
        "POL |",
        "POLSKA|",
        "POLSKA |"
    ]

    for p in prefixes:
        if name.startswith(p):
            name = name[len(p):]

    # odstránenie kvality
    name = re.sub(
        r'\b(FHD|UHD|HD|SD|4K|HEVC|H265|H264)\b',
        '',
        name
    )

    # odstránenie krajiny
    name = re.sub(
        r'[\(\[\{]\s*(PL|POL|POLAND)\s*[\)\]\}]',
        '',
        name
    )

    # všetko okrem A-Z a čísiel -> medzera
    name = re.sub(r'[^A-Z0-9]+', ' ', name)

    # viac medzier -> jedna
    name = re.sub(r'\s+', ' ', name)

    return name.strip().lower()


# ==========================================
# NAČÍTANIE EPG
# ==========================================

for channel in root.findall("channel"):

    cid = channel.attrib.get("id", "").strip()

    if not cid:
        continue

    for name in channel.findall("display-name"):

        text = (name.text or "").strip()

        key = normalize(text)

        if key:
            channels[key] = cid


print(f"Loaded {len(channels)} channel names.")


# ==========================================
# NAČÍTANIE M3U
# ==========================================

with open(
    INPUT_M3U,
    encoding="utf-8",
    errors="ignore"
) as f:
    lines = f.readlines()


out = []

matched = 0
not_matched = 0

# aby sme nevypisovali stovky rovnakých správ
shown_missing = 0


# ==========================================
# SPRACOVANIE
# ==========================================

for line in lines:

    if line.startswith("#EXTINF"):

        # ----------------------------------
        # 1. PREFERUJEME tvg-name
        # ----------------------------------

        m_name = re.search(
            r'tvg-name="([^"]*)"',
            line,
            re.IGNORECASE
        )

        if m_name:
            title = m_name.group(1).strip()

        else:
            # fallback na názov za čiarkou
            m = re.search(r',(.+)$', line)

            if m:
                title = m.group(1).strip()
            else:
                title = ""

        key = normalize(title)

        cid = None

        # ----------------------------------
        # PRESNÁ ZHODA
        # ----------------------------------

        if key in channels:

            cid = channels[key]

        else:

            # ----------------------------------
            # ČIASTOČNÁ ZHODA
            # ----------------------------------

            for epg_name, epg_id in channels.items():

                if not epg_name:
                    continue

                if key == epg_name:
                    cid = epg_id
                    break

                if key in epg_name:
                    cid = epg_id
                    break

                if epg_name in key:
                    cid = epg_id
                    break

        # ----------------------------------
        # NAŠIEL SA
        # ----------------------------------

        if cid:

            if 'tvg-id="' in line:

                line = re.sub(
                    r'tvg-id="[^"]*"',
                    f'tvg-id="{cid}"',
                    line,
                    count=1
                )

            else:

                line = line.replace(
                    "#EXTINF:-1",
                    f'#EXTINF:-1 tvg-id="{cid}"',
                    1
                )

            matched += 1

        # ----------------------------------
        # NENAŠIEL SA
        # ----------------------------------

        else:

            not_matched += 1

            if shown_missing < 30:

                print(
                    f'NOT FOUND: "{title}"'
                    f'  ->  "{key}"'
                )

                shown_missing += 1

    out.append(line)


# ==========================================
# ULOŽENIE
# ==========================================

with open(
    OUTPUT_M3U,
    "w",
    encoding="utf-8"
) as f:
    f.writelines(out)


print()
print(f"Matched:     {matched}")
print(f"Not matched: {not_matched}")
print(f"Done: {OUTPUT_M3U}")