# -*- coding: utf-8 -*-
"""
네이버 통합검색에서 아파트 단지의 확인된 정보만 보강한다.

안전 원칙
- 매물 API가 아파트이고 공식 단지명이 있을 때만 검색한다.
- 검색어에는 동/호/층/거래유형을 붙이지 않는다.
- 검색 결과에 같은 단지명의 네이버페이 부동산 영역이 없으면 사용하지 않는다.
- 주소가 양쪽에 있을 때 서로 충돌하면 사용하지 않는다.
- AI 브리핑은 원문 저장/복사 대상이 아니라 Ollama 참고용 근거 텍스트로만 보관한다.
"""

import re
import unicodedata
from datetime import datetime
from urllib.parse import quote_plus


APT_TYPE_CODES = {"APT"}
APT_TYPE_NAMES = {"아파트"}


def _text(value):
    value = unicodedata.normalize("NFKC", str(value or ""))
    return re.sub(r"\s+", " ", value).strip()


def _compact(value):
    return re.sub(r"[\s\-_·ㆍ.,()\[\]]+", "", _text(value)).lower()


def _first(*values):
    for value in values:
        value = _text(value)
        if value and value.lower() not in {"null", "none", "-", "해당없음"}:
            return value
    return ""


def _article_parts(data):
    data = data if isinstance(data, dict) else {}
    return (
        data.get("articleDetail") or {},
        data.get("articleAddition") or {},
    )


def resolve_official_complex_name(data):
    """제목 추측보다 매물 API의 공식 단지명 필드를 우선한다."""
    detail, addition = _article_parts(data)
    name = _first(
        detail.get("complexName"),
        detail.get("aptName"),
        addition.get("complexName"),
        addition.get("aptName"),
    )
    if not name:
        return ""

    # API 필드에 동이 붙은 예외만 제거한다. 제목 전체를 정규식 추측하지 않는다.
    name = re.sub(r"\s+\d{1,4}\s*동$", "", name).strip()
    return name


def get_article_address(data):
    detail, addition = _article_parts(data)
    complex_detail = data.get("complexDetail") or {}
    return _first(
        data.get("complexResolvedAddress"),
        complex_detail.get("roadAddress"),
        complex_detail.get("address"),
        detail.get("roadAddress"),
        detail.get("exposureAddress"),
        addition.get("exposureAddress"),
    )


def is_apartment_article(data):
    detail, addition = _article_parts(data)
    code = _first(
        detail.get("realestateTypeCode"),
        detail.get("articleTypeCode"),
        addition.get("realEstateTypeCode"),
        addition.get("articleRealEstateTypeCode"),
    ).upper()
    name = _first(
        detail.get("realestateTypeName"),
        addition.get("realEstateTypeName"),
        addition.get("articleRealEstateTypeName"),
    )
    return code in APT_TYPE_CODES or name in APT_TYPE_NAMES


def build_decision(data):
    complex_name = resolve_official_complex_name(data)
    if not is_apartment_article(data):
        return {
            "eligible": False,
            "verified": False,
            "reason": "not_apartment",
            "query": "",
            "complex_name": "",
        }
    if not complex_name:
        return {
            "eligible": False,
            "verified": False,
            "reason": "official_complex_name_missing",
            "query": "",
            "complex_name": "",
        }
    return {
        "eligible": True,
        "verified": False,
        "reason": "pending_search",
        "query": complex_name,
        "complex_name": complex_name,
    }


def _address_tokens(value):
    value = _text(value)
    # 시/군/구/동 및 도로명처럼 비교 가치가 있는 토큰만 사용한다.
    return {
        token
        for token in re.findall(r"[가-힣A-Za-z0-9]+(?:시|군|구|읍|면|동|로|길)|\d{1,4}", value)
        if len(token) >= 2
    }


def addresses_conflict(article_address, search_text):
    article_tokens = _address_tokens(article_address)
    search_tokens = _address_tokens(search_text)
    if not article_tokens or not search_tokens:
        return False
    locality = {x for x in article_tokens if re.search(r"(시|군|구|읍|면|동)$", x)}
    return bool(locality) and not bool(locality & search_tokens)


def _extract_assigned_schools(text):
    text = _text(text)
    marker = re.search(
        r"배정\s*학교\s*(.{0,450}?)(?=(?:학군정보|단지정보|매물정보|시세|관리비|$))",
        text,
        flags=re.IGNORECASE,
    )
    if not marker:
        return []

    block = marker.group(1)
    patterns = [
        (r"([가-힣A-Za-z0-9·ㆍ\-]{2,30}초(?:등학교)?)", "초등학교", "elementary"),
        (r"([가-힣A-Za-z0-9·ㆍ\-]{2,30}중(?:학교)?)", "중학교", "middle"),
        (r"([가-힣A-Za-z0-9·ㆍ\-]{2,30}고(?:등학교)?)(?!등학군)", "고등학교", "high"),
    ]
    result = []
    seen = set()
    for pattern, school_type, school_level in patterns:
        for match in re.finditer(pattern, block):
            name = _text(match.group(1))
            # '세종시고등학군'의 '세종시고'처럼 학군 명칭 일부를 학교로 오인하지 않는다.
            if block[match.end():match.end() + 3] == "등학군":
                continue
            key = _compact(name)
            if not key or key in seen:
                continue
            seen.add(key)
            result.append({
                "school_name": name,
                "school_type": school_type,
                "school_level": school_level,
                "distance_text": "",
                "distance_meter": 0,
                "assignment_status": "assigned",
                "source": "naver_pay_realestate",
            })
    return result[:20]


def _extract_labeled_block(text, labels, stop_labels, limit=2500):
    text = _text(text)
    label_pattern = "|".join(re.escape(x) for x in labels)
    stop_pattern = "|".join(re.escape(x) for x in stop_labels)
    match = re.search(
        rf"(?:{label_pattern})\s*(.{{0,{limit}}}?)(?=(?:{stop_pattern})|$)",
        text,
        flags=re.IGNORECASE,
    )
    return _text(match.group(1)) if match else ""


def _extract_facts(text):
    facts = {}
    rules = {
        "address": r"((?:[가-힣]+(?:특별자치시|특별시|광역시|도)\s*)?[가-힣0-9]+(?:시|군|구)\s+[가-힣0-9]+(?:로|길)\s+\d+(?:-\d+)?)",
        "households": r"(\d[\d,]*)\s*세대",
        "buildings": r"(\d+)\s*개동",
        "completion": r"((?:19|20)\d{2}[./년]\s*\d{1,2}(?:[./월]\s*\d{1,2}일?)?)",
        "parking": r"(?:주차|총주차대수)[^\d]{0,15}(\d[\d,]*)\s*대",
        "floor_area_ratio": r"용적률\s*(\d+(?:\.\d+)?)\s*%",
        "building_coverage": r"건폐율\s*(\d+(?:\.\d+)?)\s*%",
    }
    for key, pattern in rules.items():
        match = re.search(pattern, text)
        if match:
            facts[key] = _text(match.group(1))
    return facts


def _extract_nearby_facilities(text, source="naver_pay_realestate"):
    rules = [
        (r"([가-힣A-Za-z0-9·ㆍ\-]{2,35}(?:역|정류장))", "교통", "transport"),
        (r"([가-힣A-Za-z0-9·ㆍ\-]{2,35}공원)", "공원/녹지", "park"),
        (r"([가-힣A-Za-z0-9·ㆍ\-]{2,35}(?:병원|의원|약국))", "의료", "medical"),
        (r"([가-힣A-Za-z0-9·ㆍ\-]{2,35}(?:마트|백화점|시장))", "쇼핑/편의", "shopping"),
        (r"([가-힣A-Za-z0-9·ㆍ\-]{2,35}(?:도서관|주민센터|행정복지센터))", "공공/문화", "public"),
    ]
    result = []
    seen = set()
    for pattern, facility_type, category in rules:
        for name in re.findall(pattern, _text(text)):
            name = _text(name)
            key = _compact(name)
            if not key or key in seen:
                continue
            seen.add(key)
            result.append({
                "facility_name": name,
                "facility_type": facility_type,
                "facility_category": category,
                "distance_text": "",
                "distance_meter": 0,
                "source": source,
            })
    return result[:20]


def _find_naver_pay_result(page, complex_name):
    compact_name = _compact(complex_name)
    candidates = page.locator(
        "a[href*='new.land.naver.com'], "
        "a[href*='fin.land.naver.com'], "
        "a[href*='land.naver.com']"
    )
    matches = []
    for index in range(min(candidates.count(), 80)):
        link = candidates.nth(index)
        try:
            text = _text(link.inner_text(timeout=700))
            href = _text(link.get_attribute("href"))
        except Exception:
            continue
        nearby = text
        container = None
        try:
            ancestors = link.locator("xpath=ancestor::*[self::li or self::section or self::div]")
            scoped_candidates = []
            for depth in range(min(ancestors.count(), 8)):
                candidate = ancestors.nth(depth)
                candidate_text = _text(candidate.inner_text(timeout=700))
                if (
                    compact_name in _compact(candidate_text)
                    and ("네이버페이 부동산" in candidate_text or "배정 학교" in candidate_text)
                ):
                    scoped_candidates.append((len(candidate_text), candidate_text, candidate))
            if scoped_candidates:
                _, nearby, container = min(scoped_candidates, key=lambda item: item[0])
        except Exception:
            pass
        if compact_name and compact_name in _compact(nearby):
            matches.append({"text": nearby[:6000], "href": href, "container": container})
    return matches


def _expand_assigned_school_section(page, container):
    """'외 N'으로 접힌 배정학교가 있으면 단지 카드 안에서만 펼친다."""
    scope = container if container is not None else page
    selectors = [
        "button:has-text('외 ')",
        "[role='button']:has-text('외 ')",
        "button[aria-expanded='false']",
    ]
    for selector in selectors:
        try:
            buttons = scope.locator(selector)
            for index in range(min(buttons.count(), 10)):
                button = buttons.nth(index)
                label = _text(button.inner_text(timeout=500))
                aria_label = _text(button.get_attribute("aria-label"))
                nearby = _text(
                    button.locator("xpath=ancestor::*[self::li or self::div][1]").inner_text(timeout=500)
                )
                combined = " ".join((label, aria_label, nearby))
                if "배정" not in combined and not re.search(r"외\s*\d+", combined):
                    continue
                button.click(timeout=1000)
                page.wait_for_timeout(400)
        except Exception:
            continue


def _extract_ai_briefing(body_text):
    briefing = _extract_labeled_block(
        body_text,
        ["AI 브리핑", "AI Briefing"],
        ["네이버페이 부동산", "함께 많이 찾는", "관련 검색어", "뉴스"],
        limit=5000,
    )
    if not briefing:
        return {"raw_context": "", "nearby": "", "features": ""}

    nearby = _extract_labeled_block(
        briefing,
        ["주변"],
        ["특징", "단지 정보", "매물 정보"],
        limit=2200,
    )
    features = _extract_labeled_block(
        briefing,
        ["특징"],
        ["단지 정보", "매물 정보", "네이버페이 부동산"],
        limit=2200,
    )
    return {
        "raw_context": briefing[:3500],
        "nearby": nearby[:1800],
        "features": features[:1800],
    }


def collect_naver_search_enrichment(context, data, timeout_ms=20000):
    decision = build_decision(data)
    result = {
        **decision,
        "collected_at": datetime.now().isoformat(timespec="seconds"),
        "source": "naver_search",
        "sources": [],
        "complex_facts": {},
        "assigned_schools": [],
        "nearby_facilities": [],
        "ai_briefing_context": "",
        "ai_briefing_nearby": "",
        "ai_briefing_features": "",
    }
    if not decision["eligible"]:
        return result

    page = context.new_page()
    try:
        url = (
            "https://search.naver.com/search.naver"
            f"?where=nexearch&sm=top_hty&fbm=0&ie=utf8&query={quote_plus(decision['query'])}"
        )
        page.goto(url, wait_until="domcontentloaded", timeout=timeout_ms)
        page.wait_for_timeout(1200)
        body_text = _text(page.locator("body").inner_text(timeout=timeout_ms))
        pay_matches = _find_naver_pay_result(page, decision["complex_name"])

        if not pay_matches:
            result["reason"] = "matching_naver_pay_realestate_not_found"
            return result
        if addresses_conflict(get_article_address(data), body_text):
            result["reason"] = "address_conflict"
            return result

        primary_match = pay_matches[0]
        _expand_assigned_school_section(page, primary_match.get("container"))
        if primary_match.get("container") is not None:
            try:
                pay_card_text = _text(
                    primary_match["container"].inner_text(timeout=timeout_ms)
                )
            except Exception:
                pay_card_text = primary_match.get("text") or ""
        else:
            pay_card_text = primary_match.get("text") or ""

        result["verified"] = True
        result["reason"] = "verified"
        result["sources"].append({
            "type": "naver_pay_realestate",
            "url": primary_match["href"],
        })
        # 네이버페이 부동산 단지 카드 내부만 구조화 정보의 1순위 근거로 사용한다.
        result["complex_facts"] = _extract_facts(pay_card_text)
        result["assigned_schools"] = _extract_assigned_schools(pay_card_text)
        result["nearby_facilities"] = _extract_nearby_facilities(
            pay_card_text,
            source="naver_pay_realestate",
        )

        # AI 브리핑은 '주변'과 '특징'을 분리해 사실 참고 근거로만 전달한다.
        ai_briefing = _extract_ai_briefing(body_text)
        if ai_briefing["raw_context"]:
            result["ai_briefing_context"] = ai_briefing["raw_context"][:3500]
            result["ai_briefing_nearby"] = ai_briefing["nearby"]
            result["ai_briefing_features"] = ai_briefing["features"]
            result["sources"].append({
                "type": "naver_search_ai_briefing",
                "usage": "fact_reference_only",
            })
            ai_facilities = _extract_nearby_facilities(
                ai_briefing["nearby"],
                source="naver_search_ai_briefing",
            )
            existing = {_compact(x.get("facility_name")) for x in result["nearby_facilities"]}
            result["nearby_facilities"].extend(
                x for x in ai_facilities
                if _compact(x.get("facility_name")) not in existing
            )

        print("[NAVER PAY CARD FOUND]", decision["complex_name"])
        print("[NAVER PAY CARD TEXT]", pay_card_text[:1200])
        print("[NAVER PAY COMPLEX FACTS]", result["complex_facts"])
        print("[ASSIGNED SCHOOL RAW]", _extract_labeled_block(
            pay_card_text,
            ["배정 학교"],
            ["학군정보", "단지정보", "매물정보", "시세", "관리비"],
            limit=1200,
        ) or "-")
        for school in result["assigned_schools"]:
            print("[ASSIGNED SCHOOL FOUND]", school["school_name"], school["school_type"])
        print(
            "[AI BRIEFING]",
            "nearby=", bool(result["ai_briefing_nearby"]),
            "features=", bool(result["ai_briefing_features"]),
        )
        return result
    except Exception as exc:
        result["reason"] = "search_error"
        result["error"] = str(exc)[:500]
        return result
    finally:
        page.close()
