# -*- coding: utf-8 -*-

import re
import time
from urllib.parse import quote

from playwright.sync_api import sync_playwright


def uniq_keep_order(items):
    seen = set()
    result = []

    for item in items:
        item = str(item or "").strip()

        if not item or item in seen:
            continue

        seen.add(item)
        result.append(item)

    return result


def extract_article_nos(text):
    text = str(text or "")

    patterns = [
        r'"atclNo"\s*:\s*"([0-9]{8,})"',
        r'"atclNo"\s*:\s*([0-9]{8,})',
        r'"articleNo"\s*:\s*"([0-9]{8,})"',
        r'"articleNo"\s*:\s*([0-9]{8,})',
        r'"articleNumber"\s*:\s*"([0-9]{8,})"',
        r'"articleNumber"\s*:\s*([0-9]{8,})',
        r"articleNo[=:\"'\s]+([0-9]{8,})",
        r"/articles/([0-9]{8,})",
    ]

    results = []

    for pattern in patterns:
        results.extend(
            re.findall(pattern, text, flags=re.IGNORECASE)
        )

    return uniq_keep_order(results)


def build_office_urls(naver_realtor_id):
    realtor_id = quote(str(naver_realtor_id).strip())

    return [
        f"https://m.land.naver.com/agency/info/{realtor_id}",

        f"https://new.land.naver.com/offices?realtorId={realtor_id}",

        f"https://new.land.naver.com/offices?ms=2AL111,3zkSwo,16&a=APT:OPST:SG:SMS:GJCG:APTHGJ:GM:TJ&b=A1&e=RETAIL&realtorId={realtor_id}",

        f"https://new.land.naver.com/offices?ms=2AL111,3zkSwo,16&a=APT:OPST:SG:SMS:GJCG:APTHGJ:GM:TJ&b=B1&e=RETAIL&realtorId={realtor_id}",

        f"https://new.land.naver.com/offices?ms=2AL111,3zkSwo,16&a=APT:OPST:SG:SMS:GJCG:APTHGJ:GM:TJ&b=B2&e=RETAIL&realtorId={realtor_id}",
    ]


def collect_articles_from_office_page(page, url):

    network_texts = []
    network_urls = []

    def handle_request(request):

        try:
            request_url = request.url

            if (
                "/front-api/v1/realtor/articles" in request_url
                or "/api/articles" in request_url
            ):
                print("\n[ARTICLE REQUEST]")
                print(request.method, request_url)

                if request.post_data:
                    print("[POST DATA]")
                    print(request.post_data)

        except Exception as e:
            print("[HANDLE REQUEST ERROR]", e)

    page.on("request", handle_request)

    def handle_response(response):

        try:
            response_url = response.url
            status = response.status

            if response_url not in network_urls:
                network_urls.append(response_url)

            is_article_api = (
                "/api/articles" in response_url
                or "/front-api/v1/realtor/articles" in response_url
                or "/agency/info/" in response_url
            )

            if not is_article_api:
                return

            text = response.text()

            if not text:
                return

            article_nos = extract_article_nos(text)

            print("\n[ARTICLE RESPONSE]")
            print(response_url)
            print("[STATUS]", status)
            print("[TEXT LENGTH]", len(text))
            print("[ARTICLE COUNT]", len(article_nos))

            if article_nos:
                print("[ARTICLE NOS]", article_nos[:30])

                network_texts.append(text)

        except Exception as e:
            print("[HANDLE RESPONSE ERROR]", e)

    page.on("response", handle_response)

    try:

        page.goto(
            url,
            wait_until="domcontentloaded",
            timeout=60000
        )

    except Exception as e:
        print("[PAGE GOTO ERROR]", url, e)

    page.wait_for_timeout(5000)

    # 최초 강제 스크롤
    try:

        page.evaluate("""
            () => {

                window.scrollTo(
                    0,
                    document.body.scrollHeight
                );

                const els =
                    document.querySelectorAll('*');

                els.forEach(el => {

                    try {
                        el.scrollTop = el.scrollHeight;
                    } catch(e) {}

                });
            }
        """)

    except Exception:
        pass

    last_count = 0
    stable_count = 0

    # 추가 페이지 로딩 유도
    for i in range(40):

        try:

            page.mouse.wheel(0, 5000)

            page.evaluate("""
                () => {

                    window.scrollTo(
                        0,
                        document.body.scrollHeight
                    );

                    const els =
                        document.querySelectorAll('*');

                    els.forEach(el => {

                        try {
                            el.scrollTop = el.scrollHeight;
                        } catch(e) {}

                    });
                }
            """)

            page.wait_for_timeout(1800)

            html_now = page.content()

            combined_now = (
                html_now
                + "\\n"
                + "\\n".join(network_texts)
            )

            current_count = len(
                extract_article_nos(combined_now)
            )

            print(
                f"[SCROLL CHECK] "
                f"step={i+1} "
                f"articles={current_count}"
            )

            if current_count == last_count:
                stable_count += 1
            else:
                stable_count = 0

            if stable_count >= 8:
                break

            last_count = current_count

        except Exception as e:
            print("[SCROLL ERROR]", e)

    html = page.content()
    current_url = page.url

    combined = (
        html
        + "\n"
        + current_url
        + "\n"
        + "\n".join(network_texts)
    )

    article_nos = extract_article_nos(combined)

    try:

        links = page.locator("a").evaluate_all(
            "(els) => els.map(a => a.href).filter(Boolean)"
        )

        for href in links:
            article_nos.extend(
                extract_article_nos(href)
            )

    except Exception:
        links = []

    try:
        page.remove_listener(
            "response",
            handle_response
        )
    except Exception:
        pass

    try:
        page.remove_listener(
            "request",
            handle_request
        )
    except Exception:
        pass

    article_nos = uniq_keep_order(article_nos)

    return {
        "url": url,
        "final_url": current_url,
        "html_length": len(html),
        "network_response_count": len(network_texts),
        "network_text_length": sum(
            len(x)
            for x in network_texts
        ),
        "network_urls": network_urls,
        "links_count": len(links),
        "article_nos": article_nos,
    }


def find_articles_by_naver_realtor_id(
    naver_realtor_id,
    headless=True
):

    naver_realtor_id = str(
        naver_realtor_id or ""
    ).strip()

    if not naver_realtor_id:
        return {
            "ok": False,
            "error": "단체아이디가 없습니다.",
            "candidates": [],
            "debug_sources": [],
        }

    candidates = []
    debug_sources = []

    urls = build_office_urls(
        naver_realtor_id
    )

    with sync_playwright() as p:

        browser = p.chromium.launch(
            headless=headless,
            args=[
                "--disable-blink-features=AutomationControlled",
                "--no-sandbox",
            ]
        )

        context = browser.new_context(
            locale="ko-KR",
            viewport={
                "width": 1365,
                "height": 900
            },
            user_agent=(
                "Mozilla/5.0 "
                "(Windows NT 10.0; Win64; x64) "
                "AppleWebKit/537.36 "
                "(KHTML, like Gecko) "
                "Chrome/136.0.0.0 "
                "Safari/537.36"
            )
        )

        page = context.new_page()

        for url in urls:

            try:

                source = collect_articles_from_office_page(
                    page,
                    url
                )

                debug_sources.append(source)

                for article_no in source.get(
                    "article_nos",
                    []
                ):

                    candidates.append({
                        "article_no": article_no,
                        "source_keyword": naver_realtor_id,
                        "source_url":
                            source.get("final_url")
                            or url,
                        "source_status_code": 200,
                        "collector":
                            "office_realtor_id",
                    })

            except Exception as e:

                debug_sources.append({
                    "url": url,
                    "final_url": "",
                    "html_length": 0,
                    "network_response_count": 0,
                    "network_text_length": 0,
                    "article_nos": [],
                    "error": str(e),
                })

            time.sleep(1.0)

        context.close()
        browser.close()

    unique = {}

    for item in candidates:

        article_no = str(
            item.get("article_no") or ""
        ).strip()

        if (
            article_no
            and article_no not in unique
        ):
            unique[article_no] = item

    return {
        "ok": True,
        "naver_realtor_id": naver_realtor_id,
        "candidates": list(unique.values()),
        "debug_sources": debug_sources,
    }