# -*- coding: utf-8 -*-

import time
import random
import re
from datetime import datetime

from playwright.sync_api import sync_playwright

from services.fin_land_parser import parse_fin_land_article_html


HEADERS = {
    "User-Agent": (
        "Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
        "AppleWebKit/537.36 (KHTML, like Gecko) "
        "Chrome/136.0.0.0 Safari/537.36"
    )
}


def extract_first_posted_at(text):
    """
    네이버 부동산 상세 하단 예:
    최초게재 2026.05.15｜부동산뱅크 제공
    """

    if not text:
        return None

    patterns = [
        r"최초게재\s*(\d{4})\.(\d{1,2})\.(\d{1,2})",
        r"최초\s*게재\s*(\d{4})\.(\d{1,2})\.(\d{1,2})",
    ]

    for pattern in patterns:
        m = re.search(pattern, text)

        if not m:
            continue

        try:
            year = int(m.group(1))
            month = int(m.group(2))
            day = int(m.group(3))

            return datetime(year, month, day).strftime("%Y-%m-%d 00:00:00")

        except Exception:
            return None

    return None


def fetch_fin_land_article_html(article_no):
    article_no = str(article_no).strip()

    url = f"https://fin.land.naver.com/articles/{article_no}"

    time.sleep(random.uniform(0.3, 0.8))

    network_texts = []

    with sync_playwright() as p:
        browser = p.chromium.launch(
            headless=True,
            args=[
                "--disable-blink-features=AutomationControlled",
                "--no-sandbox",
            ],
        )

        context = browser.new_context(
            locale="ko-KR",
            viewport={
                "width": 1400,
                "height": 1000,
            },
            user_agent=HEADERS["User-Agent"],
        )

        page = context.new_page()

        def handle_response(response):
            try:
                response_url = response.url

                is_target = any([
                    "/article/" in response_url,
                    "/articles/" in response_url,
                    "/front-api/" in response_url,
                    "/complex/" in response_url,
                    "/photo/" in response_url,
                    "/image/" in response_url,
                ])

                if not is_target:
                    return

                text = response.text()

                if text and len(text) > 30:
                    network_texts.append(text)

            except Exception:
                pass

        page.on("response", handle_response)

        try:
            page.goto(
                url,
                wait_until="domcontentloaded",
                timeout=45000,
            )

        except Exception as e:
            browser.close()
            raise Exception(f"fin.land 페이지 접속 실패: {e}")

        page.wait_for_timeout(2500)

        for _ in range(4):
            try:
                page.mouse.wheel(0, 3500)
                page.evaluate("""
                    () => {
                        window.scrollTo(0, document.body.scrollHeight);
                    }
                """)
                page.wait_for_timeout(700)
            except Exception:
                pass

        html = page.content()
        current_url = page.url

        browser.close()

    combined = (
        html
        + "\n"
        + current_url
        + "\n"
        + "\n".join(network_texts)
    )

    if "페이지를 찾을 수 없습니다" in combined:
        raise Exception("fin.land 페이지를 찾을 수 없습니다.")

    return combined


def collect_fin_land_article_detail(article_no):
    html = fetch_fin_land_article_html(article_no)

    data = parse_fin_land_article_html(html)

    if not data.get("article_no"):
        data["article_no"] = str(article_no)

    first_posted_at = extract_first_posted_at(html)

    data["first_posted_at"] = first_posted_at

    return data