# -*- coding: utf-8 -*-

import re
import time
import random
import requests
from urllib.parse import quote


MIN_MATCH_SCORE = 30

HEADERS = {
    "User-Agent": (
        "Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
        "AppleWebKit/537.36 (KHTML, like Gecko) "
        "Chrome/124.0.0.0 Safari/537.36"
    ),
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "ko-KR,ko;q=0.9,en;q=0.8",
    "Referer": "https://new.land.naver.com/",
}


def normalize_text(value):
    value = str(value or "").strip().lower()
    value = re.sub(r"\s+", "", value)
    value = value.replace("㈜", "")
    value = value.replace("(주)", "")
    value = value.replace("주식회사", "")
    return value


def normalize_phone(value):
    return re.sub(r"[^0-9]", "", str(value or ""))


def build_realtor_keywords(realtor):
    keywords = []

    office_name = str(realtor.get("office_name") or "").strip()
    representative_name = str(realtor.get("representative_name") or "").strip()
    phone = str(realtor.get("phone") or "").strip()
    phone_digits = normalize_phone(phone)

    if office_name:
        keywords.append(office_name)

    if representative_name:
        keywords.append(representative_name)

    if phone:
        keywords.append(phone)

    if phone_digits:
        keywords.append(phone_digits)

    result = []

    for keyword in keywords:
        if keyword and keyword not in result:
            result.append(keyword)

    return result


def calc_match_score(realtor, article):
    score = 0

    realtor_office = normalize_text(realtor.get("office_name"))
    realtor_name = normalize_text(realtor.get("representative_name"))
    realtor_phone = normalize_phone(realtor.get("phone"))
    realtor_region = normalize_text(realtor.get("region"))

    article_office = normalize_text(article.get("realtor_office_name"))
    article_name = normalize_text(article.get("realtor_name"))
    article_phone = normalize_phone(article.get("realtor_phone"))
    article_region = normalize_text(article.get("region"))

    if realtor_phone and article_phone:
        if realtor_phone == article_phone:
            score += 70
        elif len(realtor_phone) >= 4 and realtor_phone[-4:] in article_phone:
            score += 35

    if realtor_office and article_office:
        if realtor_office == article_office:
            score += 50
        elif realtor_office in article_office or article_office in realtor_office:
            score += 35

    if realtor_name and article_name:
        if realtor_name == article_name:
            score += 30
        elif realtor_name in article_name or article_name in realtor_name:
            score += 20

    if realtor_region and article_region:
        if realtor_region in article_region or article_region in realtor_region:
            score += 10

    return min(score, 100)


def extract_article_numbers(text):
    article_numbers = set()

    patterns = [
        r"articleNo[=:\"'\s]+([0-9]{8,12})",
        r"article_no[=:\"'\s]+([0-9]{8,12})",
        r"articleNo=([0-9]{8,12})",
        r"articles/([0-9]{8,12})",
    ]

    for pattern in patterns:
        for match in re.findall(pattern, text or ""):
            article_numbers.add(str(match).strip())

    return list(article_numbers)


def request_text(url):
    try:
        time.sleep(random.uniform(0.7, 1.5))

        response = requests.get(
            url,
            headers=HEADERS,
            timeout=15,
        )

        if response.status_code != 200:
            print("[NAVER REQUEST FAIL]", response.status_code, url)
            return ""

        return response.text

    except Exception as e:
        print("[NAVER REQUEST ERROR]", str(e), url)
        return ""


def fetch_naver_land_candidate_articles(keyword):
    """
    안정판 후보 수집 버전.

    목적:
    - 키워드 기반으로 articleNo 후보를 찾는다.
    - 정확한 중개사 검증/상세정보는 다음 단계에서 보완한다.
    """

    encoded = quote(keyword)

    search_urls = [
        f"https://search.naver.com/search.naver?query={encoded}",
        f"https://new.land.naver.com/search?ms=37.5665,126.9780,12&a=APT:OPST:VL:SG:SMS:GJCG:GM:TJ&e=RETAIL&query={encoded}",
    ]

    article_numbers = set()

    for url in search_urls:
        html = request_text(url)

        if not html:
            continue

        found = extract_article_numbers(html)

        for article_no in found:
            article_numbers.add(article_no)

    result = []

    for article_no in article_numbers:
        result.append({
            "article_no": article_no,
            "article_name": "네이버 부동산 후보 매물",
            "trade_type": "",
            "real_estate_type": "",
            "price_text": "",
            "building_name": "",
            "floor_info": "",
            "area_info": "",
            "article_url": f"https://new.land.naver.com/offices?articleNo={article_no}",
            "realtor_office_name": "",
            "realtor_name": "",
            "realtor_phone": "",
            "region": "",
        })

    return result


def collect_articles_by_realtor(realtor, limit=None):
    keywords = build_realtor_keywords(realtor)

    print("[KEYWORDS]", keywords)

    candidates = {}

    for keyword in keywords:
        print("[SEARCH KEYWORD]", keyword)

        items = fetch_naver_land_candidate_articles(keyword)

        print("[CANDIDATE COUNT]", len(items))

        for item in items:
            article_no = str(item.get("article_no") or "").strip()

            if not article_no:
                continue

            candidates[article_no] = item

    matched_articles = []

    for article_no, article in candidates.items():
        score = calc_match_score(realtor, article)

        # 후보 수집 단계에서는 중개사 정보가 비어 있을 수 있으므로 기본 점수 부여
        if score < MIN_MATCH_SCORE:
            score = 30

        matched_articles.append({
            "article_no": article.get("article_no", ""),
            "article_name": article.get("article_name", ""),
            "trade_type": article.get("trade_type", ""),
            "real_estate_type": article.get("real_estate_type", ""),
            "price_text": article.get("price_text", ""),
            "building_name": article.get("building_name", ""),
            "floor_info": article.get("floor_info", ""),
            "area_info": article.get("area_info", ""),
            "article_url": article.get("article_url", ""),
            "matched_office_name": article.get("realtor_office_name", ""),
            "matched_representative_name": article.get("realtor_name", ""),
            "matched_phone": article.get("realtor_phone", ""),
            "match_score": score,
        })

    matched_articles.sort(key=lambda x: x["match_score"], reverse=True)

    if limit:
        matched_articles = matched_articles[:int(limit)]

    return matched_articles