# -*- coding: utf-8 -*-

import re
import time
from urllib.parse import quote
from playwright.sync_api import sync_playwright


def uniq_keep_order(items):
    seen = set()
    result = []

    for item in items:
        item = str(item or "").strip()

        if not item or item in seen:
            continue

        seen.add(item)
        result.append(item)

    return result


def extract_article_nos(text):
    text = str(text or "")

    patterns = [
        r'"articleNo"\s*:\s*"([0-9]{8,12})"',
        r'"articleNo"\s*:\s*([0-9]{8,12})',
        r'"atclNo"\s*:\s*"([0-9]{8,12})"',
        r'"atclNo"\s*:\s*([0-9]{8,12})',
        r"articleNo=([0-9]{8,12})",
    ]

    results = []

    for pattern in patterns:
        results.extend(re.findall(pattern, text, flags=re.I))

    return uniq_keep_order(results)


def build_realtor_urls(naver_realtor_id, base_url=None):
    realtor_id = quote(str(naver_realtor_id).strip())

    urls = []

    if base_url:
        urls.append(base_url)

    urls.extend([
        f"https://new.land.naver.com/offices?realtorId={realtor_id}",
        f"https://m.land.naver.com/agency/info/{realtor_id}",
    ])

    return uniq_keep_order(urls)


def find_latest_article_by_realtor_id(
    naver_realtor_id,
    base_url=None,
    headless=True,
    wait_ms=7000,
):
    naver_realtor_id = str(naver_realtor_id or "").strip()

    if not naver_realtor_id:
        return {
            "ok": False,
            "error": "naver_realtor_id 없음",
            "article_no": "",
            "candidates": [],
            "debug": [],
        }

    candidates = []
    debug = []

    urls = build_realtor_urls(
        naver_realtor_id=naver_realtor_id,
        base_url=base_url,
    )

    with sync_playwright() as p:
        browser = p.chromium.launch(
            headless=headless,
            args=[
                "--disable-blink-features=AutomationControlled",
                "--no-sandbox",
            ],
        )

        context = browser.new_context(
            locale="ko-KR",
            viewport={"width": 1365, "height": 900},
            user_agent=(
                "Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
                "AppleWebKit/537.36 (KHTML, like Gecko) "
                "Chrome/136.0.0.0 Safari/537.36"
            ),
        )

        page = context.new_page()

        network_texts = []

        def handle_response(response):
            try:
                url = response.url

                if (
                    "/api/articles" not in url
                    and "/front-api/" not in url
                    and "/realtor/articles" not in url
                    and "/agency/info/" not in url
                ):
                    return

                text = response.text()

                if not text:
                    return

                article_nos = extract_article_nos(text)

                debug.append({
                    "url": url,
                    "status": response.status,
                    "count": len(article_nos),
                    "article_nos": article_nos[:20],
                })

                if article_nos:
                    network_texts.append(text)

            except Exception as e:
                debug.append({
                    "url": getattr(response, "url", ""),
                    "error": str(e),
                })

        page.on("response", handle_response)

        for url in urls:
            try:
                print("[OPEN]", url)

                page.goto(
                    url,
                    wait_until="domcontentloaded",
                    timeout=60000,
                )

                page.wait_for_timeout(2500)

                # 최근개재일/최신순 정렬 버튼이 있으면 클릭 시도
                sort_texts = [
                    "최근개재일순",
                    "최근 개재일순",
                    "최신순",
                    "최근등록순",
                    "최신",
                ]

                for text in sort_texts:
                    try:
                        loc = page.locator(f"text={text}").first()
                        if loc.count() > 0:
                            loc.click(timeout=1500)
                            page.wait_for_timeout(2500)
                            print("[SORT CLICK]", text)
                            break
                    except Exception:
                        pass

                page.wait_for_timeout(wait_ms)

                html = page.content()
                current_url = page.url

                combined = (
                    html
                    + "\n"
                    + current_url
                    + "\n"
                    + "\n".join(network_texts)
                )

                found = extract_article_nos(combined)

                for article_no in found:
                    candidates.append({
                        "article_no": article_no,
                        "source_url": current_url,
                        "source_keyword": naver_realtor_id,
                        "collector": "latest_realtor_page",
                    })

                if candidates:
                    break

            except Exception as e:
                debug.append({
                    "url": url,
                    "error": str(e),
                })

        context.close()
        browser.close()

    unique = {}
    for item in candidates:
        article_no = str(item.get("article_no") or "").strip()
        if article_no and article_no not in unique:
            unique[article_no] = item

    candidates = list(unique.values())

    latest_article_no = candidates[0]["article_no"] if candidates else ""

    return {
        "ok": True,
        "naver_realtor_id": naver_realtor_id,
        "article_no": latest_article_no,
        "candidates": candidates,
        "debug": debug,
    }