# -*- coding: utf-8 -*-

import re
import json


def html_unescape_next_payload(text):
    if not text:
        return ""

    text = str(text)
    text = text.replace('\\"', '"')
    text = text.replace("\\/", "/")
    text = text.replace("\\\\", "\\")
    text = text.replace("\\n", "\n")
    text = text.replace("\\u0026", "&")
    text = text.replace("\\u003d", "=")
    text = text.replace("\\u003c", "<")
    text = text.replace("\\u003e", ">")
    text = text.replace("&amp;", "&")

    return text


def find_json_blocks_by_query_key(html):
    restored = html_unescape_next_payload(html)

    blocks = {}

    query_names = [
        "GET /article/key",
        "GET /article/basicInfo",
        "GET /complex",
        "GET /article/complexPrice",
        "GET /article/realPrice",
        "GET /schools",
        "GET /article/publicPrice",
    ]

    for query_name in query_names:
        pos = restored.find(query_name)

        if pos == -1:
            blocks[query_name] = ""
            continue

        start = max(0, pos - 10000)
        end = min(len(restored), pos + 18000)

        blocks[query_name] = restored[start:end]

    return blocks


def extract_value(text, pattern, default=""):
    m = re.search(pattern, text or "", flags=re.DOTALL)

    if not m:
        return default

    return m.group(1)


def extract_number(text, pattern, default=0):
    value = extract_value(text, pattern, "")

    if value == "":
        return default

    try:
        if "." in value:
            return float(value)
        return int(value)
    except Exception:
        return default


def normalize_code_name(value):
    code_map = {
        "A1": "매매",
        "B1": "전세",
        "B2": "월세",
        "A01": "아파트",
        "A02": "오피스텔",
        "C01": "원룸",
        "C02": "투룸",
        "C03": "빌라",
        "C04": "주택",
        "D02": "상가",
        "D04": "사무실",
        "D05": "건물",
        "E03": "토지",
        "SG": "상가",
        "SMS": "사무실",
    }

    value = str(value or "").strip()

    return code_map.get(value, value)


def make_price_text(price):
    try:
        price = int(price or 0)
    except Exception:
        return ""

    if price <= 0:
        return ""

    if price >= 100000000:
        eok = price // 100000000
        rest = price % 100000000

        text = f"{eok}억"

        if rest:
            text += f" {rest // 10000}만"

        return text

    return f"{price // 10000}만"


def normalize_image_url(url):
    url = str(url or "").strip()

    if not url:
        return ""

    url = url.replace("\\/", "/")
    url = url.replace("\\u0026", "&")
    url = url.replace("&amp;", "&")

    if url.startswith("//"):
        url = "https:" + url

    return url


def is_bad_image_url(url):
    lower = str(url or "").lower()

    if not lower.startswith("http"):
        return True

    bad_words = [
        "favicon",
        "logo",
        "icon",
        "sprite",
        "button",
        "common",
        "blank",
        "error",
        "apple-touch",
        "profile_default",
        "noimage",
        "default",
        "loading",
        "banner",
        "npay",
        "naverpay",
        "og_1125x570",
        "land_panel",
        "adcr",
        "veta",
        "gfp",
    ]

    if any(word in lower for word in bad_words):
        return True

    bad_exts = [
        ".js",
        ".css",
        ".html",
        ".ico",
        ".svg",
        ".woff",
        ".woff2",
        ".ttf",
    ]

    path_only = lower.split("?")[0]

    if any(path_only.endswith(ext) for ext in bad_exts):
        return True

    good_hosts = [
        "pstatic.net",
        "naver.net",
        "naver.com",
        "landthumb",
        "phinf",
    ]

    if not any(host in lower for host in good_hosts):
        return True

    return False


def guess_image_category(url):
    lower = str(url or "").lower()

    if "floor" in lower or "plan" in lower or "pyeong" in lower:
        return "floorplan"

    if "bird" in lower or "air" in lower or "vr" in lower:
        return "birdview"

    if "inside" in lower or "room" in lower or "interior" in lower:
        return "inside"

    if "realtor" in lower or "agent" in lower or "profile" in lower:
        return "realtor"

    return "building"


def extract_urls_from_json_like_text(text):
    restored = html_unescape_next_payload(text)

    urls = []

    key_patterns = [
        r'"(?:imageUrl|imageURL|imgUrl|imgURL|photoUrl|photoURL|thumbnailUrl|thumbUrl|representativeImageUrl|largeImageUrl|smallImageUrl|url)"\s*:\s*"([^"]+)"',
        r"'(?:imageUrl|imageURL|imgUrl|imgURL|photoUrl|photoURL|thumbnailUrl|thumbUrl|representativeImageUrl|largeImageUrl|smallImageUrl|url)'\s*:\s*'([^']+)'",
    ]

    for pattern in key_patterns:
        for match in re.findall(pattern, restored, flags=re.IGNORECASE):
            urls.append(match)

    direct_patterns = [
        r'https?:\\?/\\?/[^"\']+\.(?:jpg|jpeg|png|webp)(?:\?[^"\']*)?',
        r'https?:\\?/\\?/[^"\']+(?:image|photo|thumb|land|article|floor|plan|pstatic|phinf)[^"\']*',
        r'//[^"\']+\.(?:jpg|jpeg|png|webp)(?:\?[^"\']*)?',
    ]

    for pattern in direct_patterns:
        for match in re.findall(pattern, restored, flags=re.IGNORECASE):
            urls.append(match)

    return urls


def extract_fin_land_image_urls(html):
    restored = html_unescape_next_payload(html)

    raw_urls = extract_urls_from_json_like_text(restored)

    urls = []

    for raw in raw_urls:
        url = normalize_image_url(raw)

        if not url:
            continue

        if is_bad_image_url(url):
            continue

        if url not in urls:
            urls.append(url)

    result = []

    for idx, url in enumerate(urls[:30], start=1):
        category = guess_image_category(url)

        result.append({
            "image_url": url,
            "image_category": category,
            "filename_prefix": category,
            "sort_order": idx,
        })

    return result


def extract_market_price_candidates(html):
    restored = html_unescape_next_payload(html)

    candidates = []

    price_patterns = [
        r'"price"\s*:\s*([0-9]+)',
        r'"dealPrice"\s*:\s*([0-9]+)',
        r'"minPrice"\s*:\s*([0-9]+)',
        r'"maxPrice"\s*:\s*([0-9]+)',
        r'"averagePrice"\s*:\s*([0-9]+)',
    ]

    found_prices = []

    for pattern in price_patterns:
        for value in re.findall(pattern, restored):
            try:
                amount = int(value)
            except Exception:
                continue

            if amount > 0 and amount not in found_prices:
                found_prices.append(amount)

    for idx, amount in enumerate(found_prices[:30], start=1):
        candidates.append({
            "price_type": "detected_price",
            "price_amount": amount,
            "price_text": make_price_text(amount),
            "sort_order": idx,
        })

    return candidates


def extract_school_candidates(html):
    restored = html_unescape_next_payload(html)

    candidates = []

    school_names = re.findall(
        r'"schoolName"\s*:\s*"([^"]+)"',
        restored,
        flags=re.IGNORECASE
    )

    for idx, name in enumerate(school_names, start=1):
        if not name:
            continue

        item = {
            "school_name": name,
            "school_type": "",
            "distance_text": "",
            "address": "",
            "tel": "",
            "homepage": "",
            "sort_order": idx,
        }

        if item not in candidates:
            candidates.append(item)

    return candidates


def parse_fin_land_article_html(html):
    blocks = find_json_blocks_by_query_key(html)

    all_text = html_unescape_next_payload(html)

    article_no = extract_value(
        all_text,
        r'"articleNumber"\s*:\s*"([0-9]+)"'
    )

    if not article_no:
        article_no = extract_value(
            all_text,
            r"매물번호</div><div[^>]*>([0-9]+)</div>"
        )

    article_name = extract_value(
        all_text,
        r'"articleName"\s*:\s*"([^"]*)"'
    )

    article_feature = extract_value(
        all_text,
        r'"articleFeatureDescription"\s*:\s*"([^"]*)"'
    )

    cp_id = extract_value(
        all_text,
        r'"cpId"\s*:\s*"([^"]*)"'
    )

    exposure_start_date = extract_value(
        all_text,
        r'"exposureStartDate"\s*:\s*"([^"]*)"'
    )

    complex_number = extract_number(
        all_text,
        r'"complexNumber"\s*:\s*([0-9]+)'
    )

    complex_name = extract_value(
        all_text,
        r'"complexName"\s*:\s*"([^"]*)"'
    )

    pyeong_type_number = extract_number(
        all_text,
        r'"pyeongTypeNumber"\s*:\s*([0-9]+)'
    )

    building_number = extract_number(
        all_text,
        r'"buildingNumber"\s*:\s*([0-9]+)'
    )

    real_estate_type = extract_value(
        all_text,
        r'"realEstateType"\s*:\s*"([^"]*)"'
    )

    trade_type = extract_value(
        all_text,
        r'"tradeType"\s*:\s*"([^"]*)"'
    )

    legal_division_number = extract_value(
        all_text,
        r'"legalDivisionNumber"\s*:\s*"([^"]*)"'
    )

    city = extract_value(
        all_text,
        r'"city"\s*:\s*"([^"]*)"'
    )

    division = extract_value(
        all_text,
        r'"division"\s*:\s*"([^"]*)"'
    )

    sector = extract_value(
        all_text,
        r'"sector"\s*:\s*"([^"]*)"'
    )

    jibun = extract_value(
        all_text,
        r'"jibun"\s*:\s*"([^"]*)"'
    )

    road_name = extract_value(
        all_text,
        r'"roadName"\s*:\s*"([^"]*)"'
    )

    price = extract_number(
        all_text,
        r'"price"\s*:\s*([0-9]+)'
    )

    target_floor = extract_value(
        all_text,
        r'"targetFloor"\s*:\s*"([^"]*)"'
    )

    total_floor = extract_value(
        all_text,
        r'"totalFloor"\s*:\s*"([^"]*)"'
    )

    room_count = extract_number(
        all_text,
        r'"roomCount"\s*:\s*([0-9]+)'
    )

    bathroom_count = extract_number(
        all_text,
        r'"bathRoomCount"\s*:\s*([0-9]+)'
    )

    direction = extract_value(
        all_text,
        r'"direction"\s*:\s*"([^"]*)"'
    )

    direction_standard = extract_value(
        all_text,
        r'"directionStandard"\s*:\s*"([^"]*)"'
    )

    supply_space = extract_number(
        all_text,
        r'"supplySpace"\s*:\s*([0-9.]+)'
    )

    exclusive_space = extract_number(
        all_text,
        r'"exclusiveSpace"\s*:\s*([0-9.]+)'
    )

    supply_space_name = extract_value(
        all_text,
        r'"supplySpaceName"\s*:\s*"([^"]*)"'
    )

    exclusive_space_name = extract_value(
        all_text,
        r'"exclusiveSpaceName"\s*:\s*"([^"]*)"'
    )

    pyeong_area = extract_number(
        all_text,
        r'"pyeongArea"\s*:\s*([0-9.]+)'
    )

    dong_name = extract_value(
        all_text,
        r'"dongName"\s*:\s*"([^"]*)"'
    )

    management_office_contact = extract_value(
        all_text,
        r'"managementOfficeContact"\s*:\s*"([^"]*)"'
    )

    realtor_office_name = extract_value(
        all_text,
        r'"realtorOfficeName"\s*:\s*"([^"]*)"'
    )

    if not realtor_office_name:
        realtor_office_name = extract_value(
            all_text,
            r'"brokerOfficeName"\s*:\s*"([^"]*)"'
        )

    realtor_name = extract_value(
        all_text,
        r'"realtorName"\s*:\s*"([^"]*)"'
    )

    if not realtor_name:
        realtor_name = extract_value(
            all_text,
            r'"brokerName"\s*:\s*"([^"]*)"'
        )

    realtor_phone = extract_value(
        all_text,
        r'"realtorPhone"\s*:\s*"([^"]*)"'
    )

    if not realtor_phone:
        realtor_phone = extract_value(
            all_text,
            r'"officePhone"\s*:\s*"([^"]*)"'
        )

    realtor_mobile = extract_value(
        all_text,
        r'"realtorMobile"\s*:\s*"([^"]*)"'
    )

    if not realtor_mobile:
        realtor_mobile = extract_value(
            all_text,
            r'"cellPhone"\s*:\s*"([^"]*)"'
        )

    license_number = extract_value(
        all_text,
        r'"licenseNumber"\s*:\s*"([^"]*)"'
    )

    if not license_number:
        license_number = extract_value(
            all_text,
            r'"registrationNumber"\s*:\s*"([^"]*)"'
        )

    registration_number = extract_value(
        all_text,
        r'"registrationNumber"\s*:\s*"([^"]*)"'
    )

    if not registration_number:
        registration_number = license_number

    total_household_number = extract_number(
        all_text,
        r'"totalHouseholdNumber"\s*:\s*([0-9]+)'
    )

    dong_count = extract_number(
        all_text,
        r'"dongCount"\s*:\s*([0-9]+)'
    )

    construction_company = extract_value(
        all_text,
        r'"constructionCompany"\s*:\s*"([^"]*)"'
    )

    building_use = extract_value(
        all_text,
        r'"buildingUse"\s*:\s*"([^"]*)"'
    )

    use_approval_date = extract_value(
        all_text,
        r'"useApprovalDate"\s*:\s*"([^"]*)"'
    )

    total_parking_count = extract_number(
        all_text,
        r'"totalParkingCount"\s*:\s*([0-9]+)'
    )

    parking_info = ""

    if total_parking_count:
        parking_info = f"총 {total_parking_count}대"

    address_parts = [city, division, sector, jibun]
    address = " ".join([x for x in address_parts if x])

    floor_info = ""

    if target_floor or total_floor:
        floor_info = f"{target_floor}/{total_floor}층"

    area_info = ""

    if supply_space or exclusive_space:
        area_info = f"공급 {supply_space}㎡ / 전용 {exclusive_space}㎡"

    price_text = make_price_text(price)

    images = extract_fin_land_image_urls(html)

    return {
        "article_no": article_no,
        "article_name": article_name,
        "article_feature": article_feature,
        "cp_id": cp_id,
        "exposure_start_date": exposure_start_date,

        "complex_number": complex_number,
        "complex_name": complex_name,
        "pyeong_type_number": pyeong_type_number,
        "building_number": building_number,
        "dong_name": dong_name,

        "real_estate_type": real_estate_type,
        "real_estate_type_name": normalize_code_name(real_estate_type),
        "trade_type": trade_type,
        "trade_type_name": normalize_code_name(trade_type),

        "legal_division_number": legal_division_number,
        "address": address,
        "road_address": road_name,
        "city": city,
        "division": division,
        "sector": sector,
        "jibun": jibun,

        "price": price,
        "price_text": price_text,

        "floor_info": floor_info,
        "target_floor": target_floor,
        "total_floor": total_floor,
        "room_count": room_count,
        "bathroom_count": bathroom_count,
        "direction": direction,
        "direction_standard": direction_standard,

        "supply_space": supply_space,
        "exclusive_space": exclusive_space,
        "supply_space_name": supply_space_name,
        "exclusive_space_name": exclusive_space_name,
        "pyeong_area": pyeong_area,
        "area_info": area_info,

        "total_household_number": total_household_number,
        "dong_count": dong_count,
        "construction_company": construction_company,
        "building_use": building_use,
        "use_approval_date": use_approval_date,
        "parking_info": parking_info,
        "management_office_contact": management_office_contact,
        "realtor_office_name": realtor_office_name,
        "realtor_name": realtor_name,
        "realtor_phone": realtor_phone,
        "realtor_mobile": realtor_mobile,
        "license_number": license_number,
        "registration_number": registration_number,

        "images": images,
        "image_count": len(images),
        "market_prices": extract_market_price_candidates(html),
        "schools": extract_school_candidates(html),

        "raw_blocks": {
            "article_key_found": bool(blocks.get("GET /article/key")),
            "article_basic_found": bool(blocks.get("GET /article/basicInfo")),
            "complex_found": bool(blocks.get("GET /complex")),
            "complex_price_found": bool(blocks.get("GET /article/complexPrice")),
            "real_price_found": bool(blocks.get("GET /article/realPrice")),
            "schools_found": bool(blocks.get("GET /schools")),
            "public_price_found": bool(blocks.get("GET /article/publicPrice")),
        }
    }


def parse_fin_land_article_file(html_path):
    with open(html_path, "r", encoding="utf-8") as f:
        html = f.read()

    return parse_fin_land_article_html(html)