# -*- coding: utf-8 -*-

import re
import time
import random
import requests
from urllib.parse import quote


HEADERS = {
    "User-Agent": (
        "Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
        "AppleWebKit/537.36 (KHTML, like Gecko) "
        "Chrome/136.0.0.0 Safari/537.36"
    ),
    "Accept": (
        "text/html,application/xhtml+xml,application/xml;q=0.9,"
        "image/avif,image/webp,image/apng,*/*;q=0.8"
    ),
    "Accept-Language": "ko-KR,ko;q=0.9,en;q=0.8",
}


def normalize_digits(value):
    return re.sub(r"[^0-9]", "", str(value or ""))


def uniq_keep_order(items):
    seen = set()
    result = []

    for item in items:
        key = str(item)

        if not key:
            continue

        if key in seen:
            continue

        seen.add(key)
        result.append(item)

    return result


def build_search_keywords(realtor):
    """
    articleNo 직접 입력 없이 중개업소 정보만으로 검색 후보 키워드 생성.

    우선순위:
    1. 개설등록번호
    2. 중개업소명
    3. 대표전화
    4. 휴대폰
    5. 대표자명
    """

    office_name = str(realtor.get("office_name") or "").strip()
    representative_name = str(realtor.get("representative_name") or "").strip()
    license_number = str(realtor.get("license_number") or "").strip()

    office_phone = str(
        realtor.get("office_phone")
        or realtor.get("phone")
        or ""
    ).strip()

    mobile_phone = str(
        realtor.get("mobile_phone")
        or realtor.get("mobile")
        or ""
    ).strip()

    business_number = str(realtor.get("business_number") or "").strip()

    keywords = []

    if license_number:
        keywords.append(f'"{license_number}"')
        keywords.append(f'"{license_number}" "네이버 부동산"')
        keywords.append(f'"{license_number}" "매물"')

    if office_name:
        keywords.append(office_name)
        keywords.append(f"{office_name} 네이버 부동산")
        keywords.append(f"{office_name} 매물")

    if office_phone:
        keywords.append(office_phone)
        digits = normalize_digits(office_phone)

        if digits:
            keywords.append(digits)
            keywords.append(f"{digits} 네이버 부동산")

    if mobile_phone:
        keywords.append(mobile_phone)
        digits = normalize_digits(mobile_phone)

        if digits:
            keywords.append(digits)

    if representative_name and office_name:
        keywords.append(f"{office_name} {representative_name}")

    if business_number:
        keywords.append(business_number)

    return uniq_keep_order(keywords)


def extract_article_nos_from_text(text):
    """
    HTML 또는 텍스트에서 articleNo 후보 추출.
    네이버 부동산 articleNo는 보통 10자리 이상 숫자.
    """

    text = str(text or "")

    patterns = [
        r"articleNo[=:\"'\s]+([0-9]{8,})",
        r"article_no[=:\"'\s]+([0-9]{8,})",
        r"/articles/([0-9]{8,})",
        r"articleNumber[=:\"'\s]+([0-9]{8,})",
        r"매물번호[^0-9]*([0-9]{8,})",
    ]

    results = []

    for pattern in patterns:
        for match in re.findall(pattern, text, flags=re.IGNORECASE):
            results.append(match)

    # 마지막 보조: 너무 넓지만 후보 보강용
    #for match in re.findall(r"\b[0-9]{10,13}\b", text):
        # 법정동코드/전화번호/타 숫자 일부 제외는 이후 상세검증에서 걸러냄
    #    results.append(match)

    return uniq_keep_order(results)


def fetch_url(url, timeout=20):
    time.sleep(random.uniform(0.5, 1.3))

    res = requests.get(
        url,
        headers=HEADERS,
        timeout=timeout,
        allow_redirects=True
    )

    return {
        "url": url,
        "final_url": res.url,
        "status_code": res.status_code,
        "text": res.text if res.status_code == 200 else "",
        "content_type": res.headers.get("content-type", ""),
    }


def search_naver_web(keyword):
    """
    일반 네이버 검색.
    일부 환경에서는 HTML이 비어 있거나 차단될 수 있음.
    """
    q = quote(keyword)
    url = f"https://search.naver.com/search.naver?query={q}"

    return fetch_url(url)


def search_fin_land(keyword):
    """
    fin.land 검색 경로.
    """
    q = quote(keyword)
    url = f"https://fin.land.naver.com/search?keyword={q}"

    return fetch_url(url)


def search_new_land(keyword):
    """
    new.land 검색 경로.
    """
    q = quote(keyword)
    url = f"https://new.land.naver.com/search?query={q}"

    return fetch_url(url)


def find_article_candidates_by_keyword(keyword):
    sources = []

    for search_func in [
        search_naver_web,
        search_fin_land,
        search_new_land,
    ]:
        try:
            result = search_func(keyword)
            article_nos = extract_article_nos_from_text(
                result.get("text", "")
            )

            sources.append({
                "keyword": keyword,
                "url": result.get("url"),
                "final_url": result.get("final_url"),
                "status_code": result.get("status_code"),
                "content_type": result.get("content_type"),
                "html_length": len(result.get("text", "")),
                "article_nos": article_nos,
            })

        except Exception as e:
            sources.append({
                "keyword": keyword,
                "url": "",
                "final_url": "",
                "status_code": 0,
                "content_type": "",
                "html_length": 0,
                "article_nos": [],
                "error": str(e),
            })

    candidates = []

    for source in sources:
        for article_no in source.get("article_nos", []):
            candidates.append({
                "article_no": str(article_no),
                "source_keyword": keyword,
                "source_url": source.get("final_url") or source.get("url"),
                "source_status_code": source.get("status_code"),
            })

    return {
        "keyword": keyword,
        "sources": sources,
        "candidates": candidates,
    }


def find_article_candidates_for_realtor(realtor, max_keywords=8):
    keywords = build_search_keywords(realtor)[:max_keywords]

    all_candidates = []
    debug_sources = []

    for keyword in keywords:
        result = find_article_candidates_by_keyword(keyword)

        debug_sources.extend(result.get("sources", []))
        all_candidates.extend(result.get("candidates", []))

    unique = {}
    for item in all_candidates:
        article_no = str(item.get("article_no") or "").strip()

        if not article_no:
            continue

        if article_no not in unique:
            unique[article_no] = item

    return {
        "keywords": keywords,
        "candidates": list(unique.values()),
        "debug_sources": debug_sources,
    }