# -*- coding: utf-8 -*-

import time
import sys
import os
import json
import urllib.request
import urllib.parse
import pymysql
import re
import socket
import traceback
from datetime import datetime

from playwright.sync_api import sync_playwright


CURRENT_DIR = os.path.dirname(os.path.abspath(__file__))
ROOT_DIR = os.path.dirname(CURRENT_DIR)

if ROOT_DIR not in sys.path:
    sys.path.append(ROOT_DIR)

from db import get_conn
from services.naver_search_complex_enricher import collect_naver_search_enrichment


# ---------------------------------------------------------
# STEP29 OPERATION LOCK / PIPELINE LOG HELPERS
# - 기존 성공 흐름은 유지하고, DB 작업락/로그만 보강한다.
# - blog_app_locks 구조: lock_name, locked_at, expires_at, owner_token
# - JSON 타입 미지원 환경을 고려해 meta_json/context_json에는 문자열 저장
# ---------------------------------------------------------

def _safe_json_dumps(data):
    try:
        return json.dumps(data or {}, ensure_ascii=False, default=str)
    except Exception:
        return "{}"


def _worker_token(prefix):
    try:
        host = socket.gethostname()
    except Exception:
        host = "unknown-host"
    return f"{prefix}:{host}:{os.getpid()}:{datetime.now().strftime('%Y%m%d%H%M%S')}"


def acquire_db_lock(lock_name, ttl_minutes=180, owner_token=None):
    """
    DB 기반 작업락.
    - 만료된 락은 제거
    - 같은 lock_name이 살아 있으면 실행하지 않음
    - 실패해도 예외를 밖으로 던지지 않고 False 반환
    """
    owner_token = owner_token or _worker_token(lock_name)
    conn = None

    try:
        conn = get_conn()

        with conn.cursor() as cur:
            cur.execute("""
                DELETE FROM blog_app_locks
                WHERE expires_at < NOW()
            """)

            cur.execute("""
                INSERT INTO blog_app_locks
                (
                    lock_name,
                    locked_at,
                    expires_at,
                    owner_token
                )
                VALUES
                (
                    %s,
                    NOW(),
                    DATE_ADD(NOW(), INTERVAL %s MINUTE),
                    %s
                )
            """, (
                str(lock_name)[:100],
                int(ttl_minutes),
                str(owner_token)[:100],
            ))

        conn.commit()
        print(f"[DB LOCK ACQUIRED] {lock_name} owner={owner_token}")
        return True, owner_token

    except Exception as e:
        try:
            if conn:
                conn.rollback()
        except Exception:
            pass

        print(f"[DB LOCKED OR ERROR] {lock_name} / {e}")
        return False, owner_token

    finally:
        try:
            if conn:
                conn.close()
        except Exception:
            pass


def release_db_lock(lock_name, owner_token):
    conn = None

    try:
        conn = get_conn()

        with conn.cursor() as cur:
            cur.execute("""
                DELETE FROM blog_app_locks
                WHERE lock_name = %s
                  AND owner_token = %s
            """, (
                str(lock_name)[:100],
                str(owner_token)[:100],
            ))

            affected = cur.rowcount

        conn.commit()
        print(f"[DB LOCK RELEASED] {lock_name} affected={affected}")

    except Exception as e:
        print(f"[DB LOCK RELEASE ERROR] {lock_name} / {e}")

    finally:
        try:
            if conn:
                conn.close()
        except Exception:
            pass


def create_pipeline_run(run_type, message="", meta=None):
    conn = None

    try:
        conn = get_conn()

        with conn.cursor() as cur:
            cur.execute("""
                INSERT INTO blog_pipeline_runs
                (
                    run_type,
                    status,
                    started_at,
                    message,
                    meta_json,
                    created_at
                )
                VALUES
                (
                    %s,
                    'running',
                    NOW(),
                    %s,
                    %s,
                    NOW()
                )
            """, (
                str(run_type)[:50],
                str(message or "")[:2000],
                _safe_json_dumps(meta or {}),
            ))

            run_id = cur.lastrowid

        conn.commit()
        print(f"[PIPELINE RUN START] run_id={run_id}, run_type={run_type}")
        return run_id

    except Exception as e:
        print(f"[PIPELINE RUN START ERROR] {run_type} / {e}")
        return None

    finally:
        try:
            if conn:
                conn.close()
        except Exception:
            pass


def finish_pipeline_run(run_id, status, total=0, success=0, failed=0, skipped=0, message="", meta=None):
    if not run_id:
        return

    conn = None

    try:
        conn = get_conn()

        with conn.cursor() as cur:
            cur.execute("""
                UPDATE blog_pipeline_runs
                SET
                    status = %s,
                    finished_at = NOW(),
                    total_count = %s,
                    success_count = %s,
                    fail_count = %s,
                    skipped_count = %s,
                    message = %s,
                    meta_json = %s
                WHERE id = %s
            """, (
                str(status)[:30],
                int(total or 0),
                int(success or 0),
                int(failed or 0),
                int(skipped or 0),
                str(message or "")[:2000],
                _safe_json_dumps(meta or {}),
                int(run_id),
            ))

        conn.commit()
        print(f"[PIPELINE RUN FINISH] run_id={run_id}, status={status}")

    except Exception as e:
        print(f"[PIPELINE RUN FINISH ERROR] run_id={run_id} / {e}")

    finally:
        try:
            if conn:
                conn.close()
        except Exception:
            pass


def add_pipeline_log(run_id=None, level="info", step_name="", message="", article_no=None, realtor_id=None, context=None):
    conn = None

    try:
        conn = get_conn()

        with conn.cursor() as cur:
            cur.execute("""
                INSERT INTO blog_pipeline_logs
                (
                    run_id,
                    level,
                    step_name,
                    article_no,
                    realtor_id,
                    message,
                    context_json,
                    created_at
                )
                VALUES
                (
                    %s,
                    %s,
                    %s,
                    %s,
                    %s,
                    %s,
                    %s,
                    NOW()
                )
            """, (
                run_id,
                str(level or "info")[:20],
                str(step_name or "")[:100],
                str(article_no or "")[:30] if article_no else None,
                realtor_id,
                str(message or "")[:2000],
                _safe_json_dumps(context or {}),
            ))

        conn.commit()

    except Exception as e:
        print(f"[PIPELINE LOG ERROR] {step_name} / {e}")

    finally:
        try:
            if conn:
                conn.close()
        except Exception:
            pass



USER_AGENT = (
    "Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
    "AppleWebKit/537.36 (KHTML, like Gecko) "
    "Chrome/136.0.0.0 Safari/537.36"
)

IMAGE_HOST = "https://landthumb-phinf.pstatic.net"

# ---------------------------------------------------------
# IMAGE STORAGE
# ---------------------------------------------------------
# local_path는 예전 구조처럼 관리자/브라우저에서 볼 수 있는 public URL로 저장합니다.
# 실제 파일은 storage/realestate/{article_no}/{category}/ 아래 저장합니다.
#
# 예:
# file_path  = D:/honghee/blog_api/storage/realestate/2628349208/article/article_1.jpg
# local_path = http://61.32.69.107:9000/storage/realestate/2628349208/article/article_1.jpg

STORAGE_DIR = os.environ.get(
    "BLOG_STORAGE_DIR",
    os.path.join(ROOT_DIR, "storage")
)

REAL_ESTATE_STORAGE_DIR = os.path.join(
    STORAGE_DIR,
    "realestate"
)

PUBLIC_STORAGE_BASE_URL = os.environ.get(
    "PUBLIC_STORAGE_BASE_URL",
    "http://61.32.69.107:9000/storage"
).rstrip("/")




class NaverResponseArticleFetcher:

    def __init__(self, headless=True):
        self.headless = headless
        self.playwright = None
        self.browser = None
        self.context = None
        self.page = None
        self._column_cache = {}
        self.extra_floorplan_images = []
        self.extra_response_images = []
        self.complex_response_cache = {}
        self.extra_ground_gallery_images = []
        self.ground_gallery_tab_used = ""
        self.extra_ground_gallery_images = []
        self.ground_gallery_tab_used = ""

        # complex API 직접 fetch가 401이어도 브라우저가 자체 호출한 응답은 받을 수 있으므로
        # response listener에서 단지/complex JSON을 임시 보관한다.
        self.complex_response_cache = {}

    # ---------------------------------------------------------
    # DB UTILS
    # ---------------------------------------------------------

    def get_columns(self, table_name):
        if table_name in self._column_cache:
            return self._column_cache[table_name]

        conn = get_conn()

        try:
            with conn.cursor(pymysql.cursors.DictCursor) as cur:
                cur.execute(f"SHOW COLUMNS FROM {table_name}")
                rows = cur.fetchall()

            columns = set(row["Field"] for row in rows)
            self._column_cache[table_name] = columns
            return columns

        finally:
            conn.close()

    def table_exists(self, table_name):
        conn = get_conn()

        try:
            with conn.cursor(pymysql.cursors.DictCursor) as cur:
                cur.execute(
                    """
                    SELECT COUNT(*) AS cnt
                    FROM information_schema.TABLES
                    WHERE TABLE_SCHEMA = DATABASE()
                      AND TABLE_NAME = %s
                    """,
                    (table_name,)
                )
                row = cur.fetchone() or {}

            return int(row.get("cnt") or 0) > 0

        finally:
            conn.close()

    def find_school_paths(self, data):
        results = []

        keywords = [
            "school",
            "education",
            "edu",
            "초등학교",
            "중학교",
            "고등학교",
            "학교",
        ]

        def walk(obj, path="root"):
            if isinstance(obj, dict):
                for key, value in obj.items():
                    key_text = str(key)

                    if any(k.lower() in key_text.lower() for k in keywords):
                        results.append((path + "." + key_text, value))

                    if isinstance(value, str):
                        if any(k.lower() in value.lower() for k in keywords):
                            results.append((path + "." + key_text, value))

                    walk(value, path + "." + key_text)

            elif isinstance(obj, list):
                for idx, item in enumerate(obj):
                    walk(item, f"{path}[{idx}]")

        walk(data)

        print("=" * 80)
        print("[SCHOOL PATH RESULTS]", len(results))

        for path, value in results[:50]:
            print("[SCHOOL PATH]", path)

            try:
                preview = json.dumps(
                    value,
                    ensure_ascii=False
                )[:500]
            except Exception:
                preview = str(value)[:500]

            print("[SCHOOL VALUE]", preview)

        print("=" * 80)

    def update_dynamic_by_id(self, table_name, row_id, data):
        columns = self.get_columns(table_name)

        filtered = {
            key: value
            for key, value in data.items()
            if key in columns
        }

        if not filtered:
            return

        set_sql = ", ".join([
            f"{key}=%s"
            for key in filtered.keys()
        ])

        values = list(filtered.values())
        values.append(row_id)

        sql = f"""
        UPDATE {table_name}
        SET
            {set_sql},
            updated_at = NOW()
        WHERE id = %s
        """

        conn = get_conn()

        try:
            with conn.cursor() as cur:
                cur.execute(sql, values)

            conn.commit()

        finally:
            conn.close()

    def insert_dynamic(self, table_name, data, unique_update=True):
        columns = self.get_columns(table_name)

        filtered = {
            key: value
            for key, value in data.items()
            if key in columns
        }

        if not filtered:
            return

        keys = list(filtered.keys())
        placeholders = ", ".join(["%s"] * len(keys))
        column_sql = ", ".join(keys)
        values = [filtered[key] for key in keys]

        if unique_update:
            update_sql = ", ".join([
                f"{key}=VALUES({key})"
                for key in keys
                if key not in ["id", "created_at"]
            ])

            sql = f"""
            INSERT INTO {table_name}
            ({column_sql})
            VALUES ({placeholders})
            ON DUPLICATE KEY UPDATE
            {update_sql}
            """
        else:
            sql = f"""
            INSERT INTO {table_name}
            ({column_sql})
            VALUES ({placeholders})
            """

        conn = get_conn()

        try:
            with conn.cursor() as cur:
                cur.execute(sql, values)

            conn.commit()

        finally:
            conn.close()

    # ---------------------------------------------------------
    # TARGET LOAD
    # ---------------------------------------------------------

    def load_targets(self, limit=20, realtor_id=None, article_no=None):
        """
        STEP122:
        V2의 운영 편의 기능 중 --realtor-id만 V1.5에 안전 백포트한다.
        기존 수집/저장 로직은 그대로 유지하고, 대상 조회 조건만 선택적으로 좁힌다.

        STEP356:
        schedule_daily_articles.py가 blog_article_work_queue에 만든
        pending/processing 작업을 상세수집 최우선 대상으로 정렬한다.
        Queue가 없는 기존 미수집 후보는 기존 id ASC 순서를 그대로 유지한다.
        """
        conn = get_conn()

        try:
            params = []

            if article_no:
                # 명시적으로 지정한 매물은 기존 수집 상태와 관계없이 재수집한다.
                where_parts = ["BINARY article_no = BINARY %s"]
                params.append(str(article_no).strip())
            else:
                where_parts = [
                    """(
                        COALESCE(detail_collected, 0) = 0
                        OR COALESCE(collect_status, '') = 'pending'
                        OR COALESCE(article_name, '') IN (
                            '',
                            '네이버 부동산 후보 매물',
                            '네이버 부동산 현재 매물'
                        )
                      )"""
                ]

            if realtor_id is not None:
                where_parts.append("realtor_id = %s")
                params.append(int(realtor_id))

            sql = f"""
            SELECT
                id,
                realtor_id,
                article_no
            FROM blog_realtor_articles
            WHERE {" AND ".join(where_parts)}
            ORDER BY
                CASE
                    WHEN EXISTS (
                        SELECT 1
                        FROM blog_article_work_queue w
                        WHERE w.realtor_id = blog_realtor_articles.realtor_id
                          AND BINARY w.article_no = BINARY blog_realtor_articles.article_no
                          AND w.work_type = 'new_article'
                          AND w.work_status IN ('pending', 'processing')
                    ) THEN 0
                    ELSE 1
                END ASC,
                id ASC
            LIMIT %s
            """

            params.append(1 if article_no else int(limit or 20))

            with conn.cursor(pymysql.cursors.DictCursor) as cur:
                cur.execute(sql, params)
                rows = cur.fetchall() or []

            print(
                "[FETCH TARGET COUNT]",
                len(rows),
                "realtor_id=",
                realtor_id,
                "article_no=",
                article_no,
            )

            for row in rows:
                print("[FETCH TARGET]", "realtor_id=", row.get("realtor_id"), "article_no=", row.get("article_no"))

            return rows

        finally:
            conn.close()

    # ---------------------------------------------------------
    # STATUS
    # ---------------------------------------------------------

    def mark_success(self, article_id):
        conn = get_conn()

        try:
            sql = """
            UPDATE blog_realtor_articles
            SET
                detail_collected = 1,
                detail_collected_at = NOW(),
                collect_status = 'collected',
                detail_error = NULL,
                updated_at = NOW()
            WHERE id = %s
            """

            with conn.cursor() as cur:
                cur.execute(sql, (article_id,))

            conn.commit()

        finally:
            conn.close()

    def mark_failed(self, article_id, error_message):
        conn = get_conn()

        try:
            sql = """
            UPDATE blog_realtor_articles
            SET
                detail_collected = 0,
                collect_status = 'failed',
                detail_error = %s,
                updated_at = NOW()
            WHERE id = %s
            """

            with conn.cursor() as cur:
                cur.execute(
                    sql,
                    (
                        str(error_message)[:1000],
                        article_id,
                    )
                )

            conn.commit()

        finally:
            conn.close()

    # ---------------------------------------------------------
    # RESPONSE LISTENER
    # ---------------------------------------------------------

    def handle_response(self, response):
        try:
            url = str(response.url).lower()

            # STEP25 FAST:
            # 운영에 필요한 이미지/평면도/매물 관련 JSON만 감시한다.
            # school/map/poi/developmentplan 탐색성 감시는 제거한다.
            target_keywords = [
                "complex",
                "floor",
                "plan",
                "pyeong",
                "article",
                "photo",
                "photos",
                "image",
                "images",
                "media",
                "thumb",
                "facility",
            ]

            if not any(keyword in url for keyword in target_keywords):
                return

            content_type = response.headers.get("content-type", "")

            if "json" not in content_type.lower():
                return

            data = response.json()

            print("[API DETECTED]", response.url)

            # 단지주소 보강용 complex API 응답 캐시.
            # page.evaluate(fetch('/api/complexes/{hscpNo}'))는 401이 날 수 있지만,
            # 네이버 화면이 직접 호출한 같은 응답은 response listener로 정상 수신되는 경우가 있다.
            try:
                complex_match = re.search(
                    r"/api/complexes/(\d+)(?:$|\?|/overview)",
                    str(response.url),
                    flags=re.IGNORECASE,
                )

                if complex_match:
                    complex_no = str(complex_match.group(1) or "").strip()
                    if complex_no:
                        self.complex_response_cache[complex_no] = {
                            "url": response.url,
                            "data": data,
                        }
                        print("[COMPLEX RESPONSE CACHED]", complex_no, response.url)
            except Exception:
                pass

            self.extract_floorplan_from_response(data)
            self.extract_any_images_from_response(data, response.url)

        except Exception:
            pass

    def extract_floorplan_from_response(self, data):
        found = []

        def walk(obj, path=""):
            if isinstance(obj, dict):
                for key, value in obj.items():
                    key_text = str(key or "").lower()
                    next_path = f"{path}.{key_text}"

                    if isinstance(value, str):
                        if self.is_floorplan_key(next_path) and self.looks_like_image(value):
                            found.append(value)

                    elif isinstance(value, list):
                        for item in value:
                            walk(item, next_path)

                    elif isinstance(value, dict):
                        walk(value, next_path)

            elif isinstance(obj, list):
                for item in obj:
                    walk(item, path)

        walk(data)

        for url in found:
            self.extra_floorplan_images.append({
                "imageSrc": url
            })

    def extract_any_images_from_response(self, data, source_url=""):
        found = []

        def detect_category(path, value):
            text = (
                str(path or "")
                + " "
                + str(value or "")
                + " "
                + str(source_url or "")
            ).lower()

            if any(k in text for k in ["profile", "realtor", "중개", "agent"]):
                return "realtor"

            if any(k in text for k in ["floor", "plan", "pyeong", "평면", "도면", "타입"]):
                return "floorplan"

            if any(k in text for k in ["building", "complex", "apt", "exterior", "outside", "단지", "외관", "전경"]):
                return "building"

            if any(k in text for k in ["map", "지도"]):
                return "map"

            return "article"

        def walk(obj, path="root"):
            if isinstance(obj, dict):
                for key, value in obj.items():
                    key_text = str(key or "")
                    next_path = f"{path}.{key_text}"

                    if isinstance(value, str):
                        if self.looks_like_image(value):
                            image_url = self.normalize_image_url(value)
                            found.append({
                                "imageSrc": image_url,
                                "imageCategory": detect_category(next_path, value),
                                "sourcePath": next_path,
                                "sourceUrl": source_url,
                            })

                    elif isinstance(value, (dict, list)):
                        walk(value, next_path)

            elif isinstance(obj, list):
                for idx, item in enumerate(obj):
                    walk(item, f"{path}[{idx}]")

        walk(data)

        seen = set()

        for item in found:
            url = self.normalize_image_url(item.get("imageSrc"))

            if not url:
                continue

            key = url.split("?")[0].strip().lower()

            if key in seen:
                continue

            seen.add(key)
            self.extra_response_images.append(item)

        if found:
            print("[RESPONSE IMAGE FOUND]", len(found), source_url)

    def trigger_lazy_image_requests(self):
        """
        STEP25 FAST:
        매물사진/평면도 지연 로딩만 가볍게 유도한다.
        단지사진은 groundPlanGallery에서 별도로 수집한다.
        """
        try:
            self.page.wait_for_timeout(600)

            for y in [450, 900]:
                self.page.evaluate(f"window.scrollTo(0, {y})")
                self.page.wait_for_timeout(250)

            click_texts = [
                "사진",
                "매물사진",
                "평면도",
            ]

            for text in click_texts:
                try:
                    locator = self.page.locator(f"text={text}")

                    if locator.count() > 0:
                        locator.first.click(force=True, timeout=1000)
                        self.page.wait_for_timeout(600)
                except Exception:
                    pass

        except Exception as e:
            print("[LAZY IMAGE TRIGGER ERROR]", str(e))


    def page_fetch_json(self, url):
        """
        현재 네이버 부동산 페이지 세션을 그대로 사용해서 추가 API를 호출합니다.
        urllib 단독 요청보다 쿠키/헤더 영향이 적어 평면도·VR·평형 API 수집에 유리합니다.
        """

        try:
            if not url:
                return None

            if url.startswith("/"):
                fetch_url = "https://new.land.naver.com" + url
            else:
                fetch_url = url

            result = self.page.evaluate(
                """
                async (url) => {
                    const res = await fetch(url, {
                        method: 'GET',
                        credentials: 'include',
                        headers: {
                            'Accept': 'application/json, text/plain, */*'
                        }
                    });

                    const text = await res.text();

                    return {
                        status: res.status,
                        url: res.url,
                        text: text
                    };
                }
                """,
                fetch_url,
            )

            status = result.get("status")
            final_url = result.get("url") or fetch_url
            text = result.get("text") or ""

            print("[EXTRA API FETCH]", status, final_url)

            if status != 200 or not text.strip():
                return None

            try:
                data = json.loads(text)
            except Exception:
                return None

            self.extract_floorplan_from_response(data)
            self.extract_any_images_from_response(data, final_url)

            return data

        except Exception as e:
            print("[EXTRA API FETCH ERROR]", str(e))
            return None


    def find_first_key_recursive(self, obj, key_candidates):
        """
        article API 내부에서 rletNo처럼 위치가 바뀔 수 있는 값을 재귀 탐색한다.
        """
        key_candidates = [str(k).lower() for k in key_candidates]

        if isinstance(obj, dict):
            for key, value in obj.items():
                if str(key).lower() in key_candidates:
                    if value not in [None, "", 0, "0"]:
                        return value

            for value in obj.values():
                found = self.find_first_key_recursive(value, key_candidates)
                if found not in [None, "", 0, "0"]:
                    return found

        elif isinstance(obj, list):
            for item in obj:
                found = self.find_first_key_recursive(item, key_candidates)
                if found not in [None, "", 0, "0"]:
                    return found

        return ""

    def extract_rlet_no(self, data):
        """
        groundPlanGallery 호출에 필요한 rletNo 추출.

        실제 article API에서는 rletNo라는 이름이 없고,
        hscpNo가 groundPlanGallery의 rletNo 역할을 하는 경우가 있다.
        예:
        articleNo=2630399288
        hscpNo=122879
        groundPlanGallery?rletNo=122879
        """
        detail = data.get("articleDetail") or {}
        addition = data.get("articleAddition") or {}

        candidates = [
            # 1순위: 명시적 rletNo 계열
            detail.get("rletNo"),
            detail.get("rletno"),
            detail.get("rletNumber"),
            detail.get("rletNumberNo"),
            detail.get("rletNoForGallery"),
            addition.get("rletNo"),
            addition.get("rletno"),
            data.get("rletNo"),
            data.get("rletno"),

            # 2순위: 네이버 articleDetail의 단지 번호 계열
            # groundPlanGallery에서는 이 값이 rletNo 자리에 들어간다.
            detail.get("hscpNo"),
            detail.get("complexNo"),
            detail.get("complexNumber"),
            detail.get("hscpNumber"),
            data.get("hscpNo"),
            data.get("complexNo"),

            # 3순위: 재귀 탐색
            self.find_first_key_recursive(data, [
                "rletNo",
                "rletno",
                "rletNumber",
                "rletNumberNo",
                "rletNoForGallery",
                "hscpNo",
                "complexNo",
                "complexNumber",
                "hscpNumber",
            ]),
        ]

        for value in candidates:
            value = str(value or "").strip()
            if re.fullmatch(r"\d{3,20}", value):
                return value

        return ""


    def extract_rlet_no_from_current_page(self):
        """
        STEP99:
        article API에서 hscpNo/complexNo/rletNo를 얻지 못한 경우,
        현재 페이지 URL/HTML/스크립트에서 단지번호 후보를 안전하게 추출한다.
        이 함수가 없으면 일부 매물에서 fetch_ground_plan_gallery()가 AttributeError로 실패한다.
        """
        try:
            if not self.page:
                return ""

            # 1) 현재 URL에서 직접 추출
            current_url = str(self.page.url or "")
            url_patterns = [
                r"(?:complexes|complexNo|hscpNo|rletNo)[=/](\d{3,20})",
                r"(?:complexNo|hscpNo|rletNo)=([0-9]{3,20})",
                r"/complexes/([0-9]{3,20})",
            ]

            for pattern in url_patterns:
                m = re.search(pattern, current_url, flags=re.IGNORECASE)
                if m:
                    value = str(m.group(1) or "").strip()
                    if re.fullmatch(r"\d{3,20}", value):
                        print("[RLET FROM CURRENT URL]", value)
                        return value

            # 2) 페이지 내부 JS 전역값/HTML에서 추출
            try:
                candidates = self.page.evaluate(
                    """
                    () => {
                        const results = [];

                        function push(v) {
                            if (v !== undefined && v !== null) {
                                results.push(String(v));
                            }
                        }

                        try {
                            push(window.complexNo);
                            push(window.hscpNo);
                            push(window.rletNo);
                            push(window.__NEXT_DATA__ && JSON.stringify(window.__NEXT_DATA__));
                            push(document.body && document.body.innerText);
                            push(document.documentElement && document.documentElement.innerHTML);
                        } catch (e) {}

                        return results;
                    }
                    """
                )
            except Exception:
                candidates = []

            patterns = [
                r'"hscpNo"\s*:\s*"?(\d{3,20})"?',
                r'"complexNo"\s*:\s*"?(\d{3,20})"?',
                r'"rletNo"\s*:\s*"?(\d{3,20})"?',
                r'hscpNo["\']?\s*[:=]\s*["\']?(\d{3,20})',
                r'complexNo["\']?\s*[:=]\s*["\']?(\d{3,20})',
                r'rletNo["\']?\s*[:=]\s*["\']?(\d{3,20})',
                r'/complexes/(\d{3,20})',
                r'complexes\?[^"\']*complexNo=(\d{3,20})',
            ]

            for text in candidates or []:
                text = str(text or "")
                if not text:
                    continue

                for pattern in patterns:
                    m = re.search(pattern, text, flags=re.IGNORECASE)
                    if m:
                        value = str(m.group(1) or "").strip()
                        if re.fullmatch(r"\d{3,20}", value):
                            print("[RLET FROM CURRENT PAGE]", value)
                            return value

        except Exception as e:
            print("[RLET CURRENT PAGE EXTRACT ERROR]", str(e))

        return ""


    def build_ground_plan_gallery_url(self, rlet_no, ptp_id="1", build_no="6"):
        """
        사용자가 확인한 안정 URL.
        rletNo만 기본적으로 변경하고, ptpId/buildNo는 article API 값이 있으면 사용한다.
        값이 없으면 확인된 기본값 ptpId=1, buildNo=6을 유지한다.
        """
        rlet_no = str(rlet_no or "").strip()
        ptp_id = str(ptp_id or "1").strip() or "1"
        build_no = str(build_no or "6").strip() or "6"

        if not rlet_no:
            return ""

        return (
            "https://land.naver.com/info/groundPlanGallery.naver"
            f"?rletNo={urllib.parse.quote(rlet_no)}"
            f"&ptpId={urllib.parse.quote(ptp_id)}"
            "&newComplex=Y"
            "&expand=false"
            f"&buildNo={urllib.parse.quote(build_no)}"
        )


    def extract_image_urls_from_html(self, html, source_url=""):
        """
        groundPlanGallery HTML/스크립트 내부에서 실제 사진 URL을 추출한다.
        단순 document.images에서는 지도만 잡히는 경우가 있어,
        HTML 전체에서 landthumb/photoinfra 계열을 우선 추출한다.
        """
        html = str(html or "")
        if not html:
            return []

        candidates = []

        # escaped slash 복원
        variants = [
            html,
            html.replace("\\/", "/"),
            html.replace("&amp;", "&"),
        ]

        # 1) 명시적 이미지 URL
        url_patterns = [
            r'https?://[^"\'\s<>]+(?:landthumb-phinf|phinf|photoinfra)[^"\'\s<>]+',
            r'//[^"\'\s<>]+(?:landthumb-phinf|phinf|photoinfra)[^"\'\s<>]+',
            r'https?://[^"\'\s<>]+\.(?:jpg|jpeg|png|webp)(?:\?[^"\'\s<>]*)?',
            r'//[^"\'\s<>]+\.(?:jpg|jpeg|png|webp)(?:\?[^"\'\s<>]*)?',
        ]

        for source in variants:
            for pattern in url_patterns:
                for url in re.findall(pattern, source, flags=re.IGNORECASE):
                    url = self.normalize_image_url(url)
                    if url:
                        candidates.append(url)

        # 2) JSON 문자열 안의 imageSrc/imageUrl/url 값
        key_patterns = [
            r'["\'](?:imageSrc|imageUrl|imgUrl|photoUrl|url|src)["\']\s*:\s*["\']([^"\']+)["\']',
            r'(?:imageSrc|imageUrl|imgUrl|photoUrl|url|src)\s*[:=]\s*["\']([^"\']+)["\']',
        ]

        for source in variants:
            for pattern in key_patterns:
                for url in re.findall(pattern, source, flags=re.IGNORECASE):
                    url = self.normalize_image_url(url)
                    if url:
                        candidates.append(url)

        # 3) 큰따옴표 밖에서 photoinfra 파일명만 잡히는 경우 보강
        for source in variants:
            for m in re.findall(r'[^"\'\s<>]*(?:photoinfra|apt_realimage|hscp_img)[^"\'\s<>]*\.(?:jpg|jpeg|png|webp)', source, flags=re.IGNORECASE):
                url = self.normalize_image_url(m)
                if url:
                    candidates.append(url)

        seen = set()
        result = []

        for url in candidates:
            url = self.normalize_image_url(url)

            if not url:
                continue

            if self.is_bad_image_url(url):
                continue

            lowered = url.lower()

            # ground gallery는 지도/정적 이미지 제외, 실제 부동산 사진 계열만 우선 허용
            allowed_photo_hosts_or_names = [
                "landthumb-phinf.pstatic.net",
                "phinf.pstatic.net",
                "photoinfra",
                "apt_realimage",
                "hscp_img",
            ]

            if not any(x in lowered for x in allowed_photo_hosts_or_names):
                continue

            key = url.split("?")[0].strip().lower()

            if key in seen:
                continue

            seen.add(key)
            result.append(url)

        return result


    def ground_gallery_score(self, url):
        """
        groundPlanGallery 후보 중 실제 단지 사진만 고른다.
        지도/static 리소스는 강하게 제외한다.
        """
        text = str(url or "").lower()

        hard_bad = [
            "static.map",
            "staticmap",
            "simg.pstatic.net",
            "staticmap.bin",
            "caller=mw_land",
            "markers=type:",
            "og_land",
            "sprite",
            "icon",
            "logo",
        ]

        if any(x in text for x in hard_bad):
            return -999

        score = 0

        high_keywords = [
            "landthumb-phinf.pstatic.net",
            "photoinfra",
            "apt_realimage",
            "hscp_img",
            "realimage",
        ]

        mid_keywords = [
            "complex",
            "building",
            "outside",
            "exterior",
            "view",
            "garden",
            "park",
            "community",
            "playground",
            "front",
            "entrance",
        ]

        bad_keywords = [
            "floor",
            "plan",
            "pyeong",
            "평면",
            "도면",
            "type=m",
            "profile",
            "realtor",
        ]

        for keyword in high_keywords:
            if keyword in text:
                score += 50

        for keyword in mid_keywords:
            if keyword in text:
                score += 15

        for keyword in bad_keywords:
            if keyword in text:
                score -= 80

        return score


    def select_ground_gallery_images(self, urls, max_count=6):
        """
        groundPlanGallery 단지사진 후보 중 최대 6장 선별.
        V8에서는 단지사진 탭 클릭 후 보이는 이미지라면 hscp_img/photoinfra 계열을 적극 허용한다.
        """
        temp = []

        for url in urls or []:
            url = self.normalize_image_url(url)

            if not url:
                continue

            if self.is_bad_image_url(url):
                continue

            lowered = url.lower()

            allowed_photo_hosts_or_names = [
                "landthumb-phinf.pstatic.net",
                "phinf.pstatic.net",
                "photoinfra",
                "apt_realimage",
                "hscp_img",
            ]

            if not any(x in lowered for x in allowed_photo_hosts_or_names):
                continue

            score = self.ground_gallery_score(url)

            # 지도/로고 등 확실한 쓰레기만 제외
            if score <= -900:
                continue

            # 실제 단지사진 탭에서 수집된 hscp_img는 점수 보정
            if "hscp_img" in lowered or "photoinfra" in lowered:
                score += 30

            temp.append({
                "url": url,
                "score": score,
            })

        temp.sort(key=lambda x: x["score"], reverse=True)

        selected = []
        seen = set()

        print("[GROUND GALLERY CANDIDATE COUNT]", len(temp))

        for item in temp[:20]:
            print("[GROUND GALLERY CANDIDATE]", item["score"], item["url"])

        for item in temp:
            url = item["url"]
            key = url.split("?")[0].strip().lower()

            if key in seen:
                continue

            seen.add(key)
            selected.append(url)

            if len(selected) >= max_count:
                break

        return selected


    def build_floorplan_exclude_keys(self, data):
        """
        article API에서 이미 평면도로 확인된 URL은 ground_gallery에서 강제 제외한다.
        """
        detail = data.get("articleDetail") or {}
        urls = []

        for key in ["grandPlanList", "floorPlanList", "articleFloorPlanList"]:
            for img in detail.get(key) or []:
                if isinstance(img, dict):
                    url = (
                        img.get("imageSrc")
                        or img.get("imageUrl")
                        or img.get("url")
                        or ""
                    )
                    url = self.normalize_image_url(url)
                    if url:
                        urls.append(url)

        for key in ["floorPlanPhoto", "representativeFloorPlan"]:
            url = self.normalize_image_url(detail.get(key) or "")
            if url:
                urls.append(url)

        for img in self.extra_floorplan_images:
            url = self.normalize_image_url(img.get("imageSrc") or "")
            if url:
                urls.append(url)

        keys = set()
        for url in urls:
            base = url.split("?")[0].strip().lower()
            if base:
                keys.add(base)

        print("[GROUND GALLERY FLOORPLAN EXCLUDE]", len(keys))
        return keys

    def collect_gallery_page_images(self, gallery_page, source_url="", floorplan_exclude_keys=None):
        """
        현재 열린 groundPlanGallery 페이지에서 보이는 이미지 URL을 수집한다.
        주변 텍스트에는 평면도/㎡가 같이 들어갈 수 있으므로 제외 조건으로 쓰지 않고,
        평면도 URL 직접 목록만 제외한다.
        """
        floorplan_exclude_keys = floorplan_exclude_keys or set()
        collected = []

        try:
            rendered_items = gallery_page.evaluate(
                """
                () => {
                    const items = [];

                    function isVisible(el) {
                        if (!el) return false;
                        const rect = el.getBoundingClientRect();
                        const style = window.getComputedStyle(el);
                        if (!rect || rect.width < 80 || rect.height < 80) return false;
                        if (style.display === 'none' || style.visibility === 'hidden' || Number(style.opacity) === 0) return false;
                        return true;
                    }

                    Array.from(document.images || []).forEach(img => {
                        if (!isVisible(img)) return;

                        const urls = [];
                        if (img.src) urls.push(img.src);

                        ['data-src','data-original','data-url','data-image','data-image-url','data-lazy-src'].forEach(attr => {
                            const v = img.getAttribute(attr);
                            if (v) urls.push(v);
                        });

                        if (img.dataset) {
                            Object.values(img.dataset).forEach(v => {
                                if (v) urls.push(v);
                            });
                        }

                        urls.forEach(url => {
                            const rect = img.getBoundingClientRect();
                            items.push({
                                url,
                                width: rect.width,
                                height: rect.height
                            });
                        });
                    });

                    Array.from(document.querySelectorAll('*')).forEach(el => {
                        if (!isVisible(el)) return;

                        const style = window.getComputedStyle(el);
                        const bg = style && style.backgroundImage;

                        if (bg && bg.includes('url(')) {
                            const m = bg.match(/url\\(["']?(.*?)["']?\\)/);
                            if (m && m[1]) {
                                const rect = el.getBoundingClientRect();
                                items.push({
                                    url: m[1],
                                    width: rect.width,
                                    height: rect.height
                                });
                            }
                        }
                    });

                    return items;
                }
                """
            )

            for item in rendered_items or []:
                image_url = self.normalize_image_url(item.get("url") or "")

                if not image_url:
                    continue

                if self.is_bad_image_url(image_url):
                    continue

                base = image_url.split("?")[0].strip().lower()
                if base in floorplan_exclude_keys:
                    print("[GROUND GALLERY SKIP FLOORPLAN URL]", image_url)
                    continue

                lowered = image_url.lower()
                allowed_photo_hosts_or_names = [
                    "landthumb-phinf.pstatic.net",
                    "phinf.pstatic.net",
                    "photoinfra",
                    "apt_realimage",
                    "hscp_img",
                ]

                if not any(x in lowered for x in allowed_photo_hosts_or_names):
                    continue

                collected.append(image_url)

        except Exception as e:
            print("[GROUND GALLERY COLLECT DOM ERROR]", str(e))

        result = []
        seen = set()

        for image_url in collected:
            image_url = self.normalize_image_url(image_url)

            if not image_url or self.is_bad_image_url(image_url):
                continue

            base = image_url.split("?")[0].strip().lower()

            if base in floorplan_exclude_keys:
                continue

            if base in seen:
                continue

            seen.add(base)
            result.append(image_url)

        return result

    def click_gallery_tab_by_category(self, gallery_page, category_code, label):
        """
        groundPlanGallery 내부 탭을 category_code 기준으로 클릭한다.
        """
        print(f"[GROUND GALLERY TAB {category_code} TRY] {label}")

        selectors = [
            f'li[_categorycode="{category_code}"] a',
            f'a[_categorycode="{category_code}"]',
            f'li[_categorycode="{category_code}"]',
            f'._js_category_item[_categorycode="{category_code}"] a',
            f'._js_category_item[_categorycode="{category_code}"]',
        ]

        for selector in selectors:
            try:
                loc = gallery_page.locator(selector)
                count = loc.count()

                if count <= 0:
                    continue

                print(f"[GROUND GALLERY TAB {category_code} SELECTOR] {selector}, count={count}")
                loc.first.click(force=True, timeout=2500)
                gallery_page.wait_for_timeout(800)

                print(f"[GROUND GALLERY TAB CLICKED] code={category_code}, label={label}")
                self.ground_gallery_tab_used = category_code
                return True

            except Exception as e:
                print(f"[GROUND GALLERY TAB {category_code} ERROR] selector={selector}, error={e}")

        try:
            clicked = gallery_page.evaluate(
                """
                ([categoryCode]) => {
                    const target =
                        document.querySelector(`li[_categorycode="${categoryCode}"] a`) ||
                        document.querySelector(`a[_categorycode="${categoryCode}"]`) ||
                        document.querySelector(`li[_categorycode="${categoryCode}"]`);

                    if (target) {
                        target.click();
                        return {
                            ok: true,
                            tag: target.tagName,
                            text: (target.innerText || '').trim(),
                            cls: target.className || ''
                        };
                    }

                    return {ok:false};
                }
                """,
                category_code
            )

            print(f"[GROUND GALLERY TAB {category_code} JS RESULT]", clicked)

            if clicked and clicked.get("ok"):
                gallery_page.wait_for_timeout(800)
                self.ground_gallery_tab_used = category_code
                return True

        except Exception as e:
            print(f"[GROUND GALLERY TAB {category_code} JS ERROR]", str(e))

        for selector in [f"text={label}", f"a:has-text('{label}')", f"li:has-text('{label}')"]:
            try:
                loc = gallery_page.locator(selector)
                count = loc.count()

                if count <= 0:
                    continue

                print(f"[GROUND GALLERY TAB TEXT TRY] selector={selector}, count={count}")
                loc.first.click(force=True, timeout=2500)
                gallery_page.wait_for_timeout(800)
                print(f"[GROUND GALLERY TAB CLICKED] code={category_code}, label={label}, selector={selector}")
                self.ground_gallery_tab_used = category_code
                return True

            except Exception as e:
                print(f"[GROUND GALLERY TAB TEXT ERROR] selector={selector}, error={e}")

        print(f"[GROUND GALLERY TAB NOT FOUND] code={category_code}, label={label}")
        return False


    def click_complex_photo_tab(self, gallery_page):
        """
        STEP25 정책:
        1. 단지사진 NEWHSCP 우선
        2. 단지사진이 없으면 단지시설 NEWHSCPFCTS 사용
        3. 둘 다 없으면 skip
        """
        self.ground_gallery_tab_used = ""

        if self.click_gallery_tab_by_category(gallery_page, "NEWHSCP", "단지사진"):
            return True

        print("[GROUND GALLERY TAB FALLBACK] NEWHSCP missing -> NEWHSCPFCTS")

        if self.click_gallery_tab_by_category(gallery_page, "NEWHSCPFCTS", "단지시설"):
            return True

        print("[GROUND GALLERY TAB NOT FOUND] NEWHSCP/NEWHSCPFCTS")
        return False


    def count_unique_good_gallery_urls(self, urls, floorplan_exclude_keys=None):
        """
        STEP100:
        ground gallery에서 실제로 저장 가능한 고유 이미지 수를 빠르게 계산한다.
        목표 6장 도달 시 썸네일/다음/키보드 탐색을 즉시 중단하기 위한 헬퍼.
        """
        floorplan_exclude_keys = floorplan_exclude_keys or set()
        seen = set()

        for image_url in urls or []:
            image_url = self.normalize_image_url(image_url)

            if not image_url:
                continue

            if self.is_bad_image_url(image_url):
                continue

            base = image_url.split("?")[0].strip().lower()

            if base in floorplan_exclude_keys:
                continue

            lowered = image_url.lower()
            allowed_photo_hosts_or_names = [
                "landthumb-phinf.pstatic.net",
                "phinf.pstatic.net",
                "photoinfra",
                "apt_realimage",
                "hscp_img",
            ]

            if not any(x in lowered for x in allowed_photo_hosts_or_names):
                continue

            seen.add(base)

        return len(seen)


    def click_gallery_thumbnails_and_collect(self, gallery_page, source_url="", floorplan_exclude_keys=None):
        """
        STEP100:
        단지사진 6장 확보 즉시 종료한다.
        이전 버전은 selector별로 최대 6개씩 반복하면서 실제 클릭이 30회 이상 발생했다.
        운영에서는 입지사진 4~6장 확보가 목적이므로 6장 도달 시 추가 탐색을 멈춘다.
        """
        floorplan_exclude_keys = floorplan_exclude_keys or set()
        target_count = 6
        all_urls = []

        def target_reached(stage):
            cnt = self.count_unique_good_gallery_urls(all_urls, floorplan_exclude_keys)
            if cnt >= target_count:
                print(f"[GROUND GALLERY TARGET REACHED] stage={stage}, count={cnt}")
                return True
            return False

        if not self.click_complex_photo_tab(gallery_page):
            print("[GROUND GALLERY COLLECT SKIP] 단지사진/단지시설 탭 없음")
            return []

        gallery_page.wait_for_timeout(800)

        all_urls.extend(self.collect_gallery_page_images(gallery_page, source_url, floorplan_exclude_keys))

        if target_reached("initial_dom"):
            return self.select_ground_gallery_images(all_urls, max_count=target_count)

        try:
            current_urls = gallery_page.evaluate(
                """
                () => {
                    const urls = [];
                    const imgs = Array.from(document.querySelectorAll('img'));

                    imgs.forEach(img => {
                        const rect = img.getBoundingClientRect();
                        const style = window.getComputedStyle(img);
                        if (!rect || rect.width < 100 || rect.height < 70) return;
                        if (style.display === 'none' || style.visibility === 'hidden') return;

                        if (img.src) urls.push(img.src);
                        ['data-src','data-original','data-url','data-image','data-image-url','data-lazy-src'].forEach(attr => {
                            const v = img.getAttribute(attr);
                            if (v) urls.push(v);
                        });
                    });

                    return urls;
                }
                """
            )

            print("[GROUND GALLERY CURRENT BIG IMG URLS]", len(current_urls or []))

            for u in current_urls or []:
                u = self.normalize_image_url(u)
                if u and not self.is_bad_image_url(u):
                    all_urls.append(u)

            if target_reached("current_big_images"):
                return self.select_ground_gallery_images(all_urls, max_count=target_count)

        except Exception as e:
            print("[GROUND GALLERY CURRENT BIG IMG ERROR]", str(e))

        for y in [300, 800]:
            try:
                gallery_page.evaluate(f"window.scrollTo(0, {y})")
                gallery_page.wait_for_timeout(350)
                all_urls.extend(self.collect_gallery_page_images(gallery_page, source_url, floorplan_exclude_keys))

                if target_reached(f"scroll_{y}"):
                    return self.select_ground_gallery_images(all_urls, max_count=target_count)

            except Exception:
                pass

        thumbnail_selectors = [
            "img:visible",
            "li:visible img",
            "a:visible img",
            "[class*='thumb']:visible",
            "[class*='thumbnail']:visible",
            "[class*='photo']:visible",
            "[class*='gallery'] li:visible",
            "[class*='swiper'] img:visible",
        ]

        clicked = 0
        global_clicked_bases = set()

        for selector in thumbnail_selectors:
            if target_reached(f"before_selector_{selector}"):
                break

            try:
                items = gallery_page.locator(selector)
                count = min(items.count(), 6)

                if count <= 0:
                    continue

                print(f"[GROUND GALLERY THUMB SELECTOR] {selector}, count={count}")

                for i in range(count):
                    if target_reached(f"thumb_loop_{selector}_{i}"):
                        break

                    try:
                        item = items.nth(i)

                        # 같은 이미지/요소를 selector별로 중복 클릭하지 않기 위한 빠른 키
                        click_key = ""
                        try:
                            click_key = item.evaluate(
                                """
                                (el) => {
                                    const img = el.tagName && el.tagName.toLowerCase() === 'img'
                                        ? el
                                        : el.querySelector && el.querySelector('img');

                                    if (img) {
                                        return img.currentSrc || img.src || img.getAttribute('data-src') || img.getAttribute('data-original') || '';
                                    }

                                    return (el.innerText || el.textContent || el.className || '').toString().slice(0, 120);
                                }
                                """
                            ) or ""
                        except Exception:
                            click_key = f"{selector}:{i}"

                        click_key = str(click_key or "").split("?")[0].strip().lower()

                        if click_key and click_key in global_clicked_bases:
                            continue

                        if click_key:
                            global_clicked_bases.add(click_key)

                        item.click(force=True, timeout=1000)
                        gallery_page.wait_for_timeout(350)

                        clicked += 1
                        all_urls.extend(self.collect_gallery_page_images(gallery_page, source_url, floorplan_exclude_keys))

                        if target_reached(f"after_click_{clicked}"):
                            break

                    except Exception:
                        continue

                if target_reached(f"after_selector_{selector}"):
                    break

            except Exception as e:
                print(f"[GROUND GALLERY THUMB ERROR] selector={selector}, error={e}")

        print("[GROUND GALLERY THUMB CLICKED]", clicked)

        if not target_reached("before_next"):
            next_selectors = [
                "button:has-text('다음')",
                "a:has-text('다음')",
                "[aria-label*='다음']",
                "[title*='다음']",
                "button[class*='next']",
                "a[class*='next']",
                ".swiper-button-next",
                "[class*='next']",
            ]

            next_clicked = 0

            for step in range(2):
                if target_reached(f"next_loop_{step}"):
                    break

                moved = False

                for selector in next_selectors:
                    try:
                        loc = gallery_page.locator(selector).first

                        if loc.count() <= 0:
                            continue

                        loc.click(force=True, timeout=1000)
                        gallery_page.wait_for_timeout(450)

                        next_clicked += 1
                        moved = True
                        all_urls.extend(self.collect_gallery_page_images(gallery_page, source_url, floorplan_exclude_keys))

                        if target_reached(f"after_next_{next_clicked}"):
                            break

                        break

                    except Exception:
                        continue

                if not moved or target_reached(f"after_next_step_{step}"):
                    break

            print("[GROUND GALLERY NEXT CLICKED]", next_clicked)
        else:
            print("[GROUND GALLERY NEXT SKIP] target reached")

        if not target_reached("before_arrow"):
            for i in range(2):
                if target_reached(f"arrow_loop_{i}"):
                    break

                try:
                    gallery_page.keyboard.press("ArrowRight")
                    gallery_page.wait_for_timeout(350)
                    all_urls.extend(self.collect_gallery_page_images(gallery_page, source_url, floorplan_exclude_keys))
                except Exception:
                    break
        else:
            print("[GROUND GALLERY ARROW SKIP] target reached")

        seen = set()
        unique_urls = []

        for image_url in all_urls:
            image_url = self.normalize_image_url(image_url)

            if not image_url:
                continue

            if self.is_bad_image_url(image_url):
                continue

            base = image_url.split("?")[0].strip().lower()
            if base in floorplan_exclude_keys:
                print("[GROUND GALLERY FINAL SKIP FLOORPLAN]", image_url)
                continue

            if base in seen:
                continue

            seen.add(base)
            unique_urls.append(image_url)

        print("[GROUND GALLERY CLICK COLLECTED]", len(unique_urls))
        return self.select_ground_gallery_images(unique_urls, max_count=target_count)


    def fetch_ground_plan_gallery(self, data):
        """
        rletNo/hscpNo 기반 groundPlanGallery 사진 수집 V9.
        """
        rlet_no = self.extract_rlet_no(data)

        if not rlet_no:
            rlet_no = self.extract_rlet_no_from_current_page()

        detail = data.get("articleDetail") or {}
        print("[RLET NO]", rlet_no)
        print("[RLET SOURCE CHECK] hscpNo=", detail.get("hscpNo"), "complexNo=", detail.get("complexNo"), "buildNo=", detail.get("buildNo"), "ptpNo=", detail.get("ptpNo"))

        if not rlet_no:
            print("[GROUND GALLERY SKIP] rletNo empty")
            return []

        floorplan_exclude_keys = self.build_floorplan_exclude_keys(data)

        url = self.build_ground_plan_gallery_url(
            rlet_no,
            ptp_id=detail.get("ptpNo") or "1",
            build_no=detail.get("buildNo") or "6",
        )
        print("[GROUND GALLERY URL]", url)

        all_urls = []
        gallery_page = None

        try:
            gallery_page = self.context.new_page()
            gallery_page.goto(url, wait_until="domcontentloaded", timeout=30000)
            gallery_page.wait_for_timeout(1500)

            click_urls = self.click_gallery_thumbnails_and_collect(
                gallery_page,
                url,
                floorplan_exclude_keys=floorplan_exclude_keys,
            )
            all_urls.extend(click_urls)

        except Exception as e:
            print("[GROUND GALLERY RENDER ERROR]", str(e))

        finally:
            try:
                if gallery_page:
                    gallery_page.close()
            except Exception:
                pass

        seen = set()
        unique_urls = []

        for image_url in all_urls:
            image_url = self.normalize_image_url(image_url)

            if not image_url:
                continue

            if self.is_bad_image_url(image_url):
                continue

            base = image_url.split("?")[0].strip().lower()
            if base in floorplan_exclude_keys:
                print("[GROUND GALLERY FINAL SKIP FLOORPLAN]", image_url)
                continue

            lowered = image_url.lower()
            allowed_photo_hosts_or_names = [
                "landthumb-phinf.pstatic.net",
                "phinf.pstatic.net",
                "photoinfra",
                "apt_realimage",
                "hscp_img",
            ]

            if not any(x in lowered for x in allowed_photo_hosts_or_names):
                continue

            if base in seen:
                continue

            seen.add(base)
            unique_urls.append(image_url)

        selected = self.select_ground_gallery_images(unique_urls, max_count=6)

        self.extra_ground_gallery_images = []

        for idx, image_url in enumerate(selected):
            self.extra_ground_gallery_images.append({
                "imageSrc": image_url,
                "imageCategory": "ground_gallery",
                "sourceUrl": url,
                "rletNo": rlet_no,
                "imageOrder": idx + 1,
            })

        print("[GROUND GALLERY FOUND]", len(unique_urls))
        print("[GROUND GALLERY SELECTED]", len(self.extra_ground_gallery_images))

        if not self.extra_ground_gallery_images:
            print("[GROUND GALLERY EMPTY] 실제 단지사진을 찾지 못해 단지사진을 저장하지 않습니다.")

        for item in self.extra_ground_gallery_images:
            print("[GROUND GALLERY IMAGE]", item.get("imageSrc"))

        return self.extra_ground_gallery_images



    def fetch_complex_address_detail(self, data):
        """
        법정 표시광고 소재지 보강용 단지/복합주소 수집.
        articleDetail.exposureAddress만 있으면 '연산동'까지만 내려오는 경우가 있어
        hscpNo 기준으로 네이버 단지 API를 추가 조회하고 raw_json에 complexDetail로 병합한다.
        별도 테이블을 만들지 않고 blog_realtor_articles.raw_json 내부에 함께 저장한다.
        """
        detail = data.get("articleDetail") or {}

        hscp_no = (
            detail.get("hscpNo")
            or detail.get("complexNo")
            or detail.get("complexNumber")
            or ""
        )

        hscp_no = str(hscp_no or "").strip()

        if not hscp_no:
            print("[COMPLEX ADDRESS SKIP] hscpNo empty")
            return data

        urls = [
            f"/api/complexes/{urllib.parse.quote(hscp_no)}",
            f"/api/complexes/{urllib.parse.quote(hscp_no)}/overview",
        ]

        found_items = []

        for url in urls:
            complex_data = self.page_fetch_json(url)
            self.page.wait_for_timeout(250)

            # 직접 fetch가 401이면, response listener가 이미 캐시한 동일 complex 응답을 사용한다.
            if not isinstance(complex_data, dict) or not complex_data:
                cached = (self.complex_response_cache or {}).get(hscp_no) or {}
                complex_data = cached.get("data") or {}
                if isinstance(complex_data, dict) and complex_data:
                    url = cached.get("url") or url
                    print("[COMPLEX ADDRESS FROM RESPONSE CACHE]", hscp_no, url)

            if not isinstance(complex_data, dict) or not complex_data:
                continue

            found_items.append({
                "url": url,
                "data": complex_data,
            })

            address_preview = self.find_complex_address_text(complex_data)
            print("[COMPLEX ADDRESS API FOUND]", url, address_preview)

        if found_items:
            # 가장 먼저 성공한 응답을 대표 complexDetail로 저장하고,
            # 나머지는 complexExtraDetails에 보관한다.
            data["complexDetail"] = found_items[0].get("data") or {}
            data["complexDetailSourceUrl"] = found_items[0].get("url") or ""

            if len(found_items) > 1:
                data["complexExtraDetails"] = found_items[1:]

            resolved_address = self.find_complex_address_text(data.get("complexDetail") or {})
            if resolved_address:
                data["complexResolvedAddress"] = resolved_address

            print("[COMPLEX ADDRESS MERGED]", hscp_no, resolved_address or "address empty")
        else:
            print("[COMPLEX ADDRESS EMPTY]", hscp_no)

        return data


    def find_complex_address_text(self, obj):
        """
        complex API 응답 구조가 바뀌어도 주소로 보이는 값을 최대한 찾는다.
        """
        address_keys = {
            "address",
            "jibunaddress",
            "jibunaddressname",
            "roadaddress",
            "roadaddressname",
            "locationaddress",
            "detailaddress",
            "exposureaddress",
            "hscpaddress",
            "complexaddress",
            "fulladdress",
            "addr",
            "jibunaddr",
            "roadaddr",
        }

        candidates = []

        def walk(value, path=""):
            if isinstance(value, dict):
                for key, child in value.items():
                    key_l = str(key or "").lower()
                    next_path = f"{path}.{key_l}" if path else key_l

                    if key_l in address_keys and isinstance(child, str):
                        text = child.strip()
                        if text:
                            candidates.append((next_path, text))

                    walk(child, next_path)

            elif isinstance(value, list):
                for idx, child in enumerate(value):
                    walk(child, f"{path}[{idx}]")

        walk(obj)

        # 주소처럼 보이고 지번이 포함된 값을 우선 사용
        def score(item):
            path, text = item
            s = 0
            if any(x in text for x in ["시", "군", "구", "동", "읍", "면"]):
                s += 10
            if re.search(r"\d", text):
                s += 20
            if "road" in path:
                s += 5
            if "jibun" in path:
                s += 8
            if "address" in path:
                s += 3
            return s

        candidates.sort(key=score, reverse=True)

        if candidates:
            return candidates[0][1]

        return ""


    def has_existing_floorplan_images(self, data):
        """
        article API 또는 response listener에서 이미 평면도가 확보되었는지 확인한다.
        확보되어 있으면 추가 API 호출을 건너뛰어 속도를 줄인다.
        """
        detail = data.get("articleDetail") or {}

        for key in ["grandPlanList", "floorPlanList", "articleFloorPlanList"]:
            if detail.get(key):
                return True

        if detail.get("floorPlanPhoto") or detail.get("representativeFloorPlan"):
            return True

        if self.extra_floorplan_images:
            return True

        return False


    def fetch_floorplan_extra_apis(self, data):
        """
        STEP25 FAST:
        article API에 평면도가 없을 때만 최소 API를 추가 호출한다.

        유지:
        - vr/representative
        - buildings/pyeongtype

        제거:
        - /api/complexes/{hscp_no}
        - /api/complexes/{hscp_no}/overview
        - 기타 시세/실거래/개발계획성 탐색
        """

        if self.has_existing_floorplan_images(data):
            print("[FLOORPLAN EXTRA API SKIP] already exists")
            return

        detail = data.get("articleDetail") or {}

        hscp_no = (
            detail.get("hscpNo")
            or detail.get("complexNo")
            or detail.get("complexNumber")
            or ""
        )

        pyeong_type_number = (
            detail.get("ptpNo")
            or detail.get("ptpName")
            or detail.get("pyeongTypeNumber")
            or ""
        )

        dong_no = (
            detail.get("buildNo")
            or detail.get("dongNo")
            or detail.get("buildingNo")
            or ""
        )

        urls = []

        if hscp_no and pyeong_type_number:
            urls.append(
                f"/api/property/complex/{hscp_no}/vr/representative"
                f"?pyeongTypeNumber={urllib.parse.quote(str(pyeong_type_number))}"
            )

        if hscp_no and dong_no:
            urls.append(
                f"/api/complexes/{hscp_no}/buildings/pyeongtype"
                f"?dongNo={urllib.parse.quote(str(dong_no))}"
            )

        seen = set()

        for url in urls[:2]:
            if url in seen:
                continue

            seen.add(url)
            self.page_fetch_json(url)
            self.page.wait_for_timeout(250)

        print(
            "[FLOORPLAN EXTRA API DONE]",
            "hscp_no=", hscp_no,
            "pyeong_type_number=", pyeong_type_number,
            "dong_no=", dong_no,
            "extra_floorplan=", len(self.extra_floorplan_images),
            "extra_response=", len(self.extra_response_images),
        )


    def is_floorplan_key(self, text):
        text = str(text or "").lower()

        keywords = [
            "floor",
            "plan",
            "grandplan",
            "pyeong",
            "평면",
            "도면",
            "type",
        ]

        return any(keyword in text for keyword in keywords)

    def looks_like_image(self, value):
        value = str(value or "").lower()

        if not value:
            return False

        if self.is_bad_image_url(value):
            return False

        image_exts = [
            ".jpg",
            ".jpeg",
            ".png",
            ".webp",
        ]

        if any(ext in value for ext in image_exts):
            return True

        image_hosts = [
            "landthumb-phinf.pstatic.net",
            "landthumb-phinf.pstatic.net",
            "phinf.pstatic.net",
            "landthumb",
        ]

        if any(host in value for host in image_hosts):
            if not any(bad in value for bad in [".svg", "sprite", "icon"]):
                return True

        return False

    # ---------------------------------------------------------
    # PARSE
    # ---------------------------------------------------------

    def normalize_image_url(self, url):
        url = str(url or "").strip()

        if not url:
            return ""

        if url.startswith("http://") or url.startswith("https://"):
            return url

        if url.startswith("//"):
            return "https:" + url

        if url.startswith("/"):
            return IMAGE_HOST + url

        return url

    def is_bad_image_url(self, url):
        """
        네이버 블로그 본문에 넣으면 깨지거나 배너로 보이는 이미지를 제외합니다.
        groundPlanGallery 테스트 중 섞였던 og_land, thmb_detail, 작은 썸네일 등을 차단합니다.
        """
        url = str(url or "").strip().lower()

        if not url:
            return True

        bad_keywords = [
            "ssl.pstatic.net/static/m/land/img/og_land",
            "ssl.pstatic.net/static.land/static/service",
            "thmb_detail",
            "sprite",
            "icon",
            "logo",
            "blank",
            "noimage",
            "no_image",

            # groundPlanGallery에서 섞이는 지도/정적 리소스 차단
            "static.map",
            "staticmap",
            "simg.pstatic.net/static.map",
            "map/staticmap",
            "staticmap.bin",
            "caller=mw_land",
            "markers=type:",
        ]

        if any(x in url for x in bad_keywords):
            return True

        # 작은 썸네일은 네이버 블로그 붙여넣기 시 깨지거나 너무 작게 보일 수 있어 제외
        small_types = [
            "type=m70", "type=m95", "type=m100", "type=m120",
            "type=m140", "type=m196", "type=m200", "type=m240",
        ]

        if any(x in url for x in small_types):
            return True

        # 잘못 붙은 URL 예: .jpgtype=m196
        if "jpgtype=" in url or "jpegtype=" in url or "pngtype=" in url:
            return True

        # GIF는 본문에서 안정성이 떨어져 제외
        if ".gif" in url:
            return True

        return False

    def image_priority_score(self, category, image_type, img):
        raw_text = json.dumps(img, ensure_ascii=False).lower()

        score = 0

        if category == "article":
            score += 200

        if category == "ground_gallery":
            score += 180

        if category == "floorplan":
            score += 120

        if category == "building":
            score += 40

        if category == "realtor":
            score += 10

        keywords_high = [
            "거실", "주방", "방", "침실", "화장실",
            "욕실", "현관", "발코니", "베란다", "뷰",
            "내부", "실내", "전경"
        ]

        keywords_mid = [
            "외관",
            "건물",
            "단지",
            "조감",
            "배치",
            "평면",
            "도면",
            "타입",
        ]

        for keyword in keywords_high:
            if keyword in raw_text:
                score += 30

        for keyword in keywords_mid:
            if keyword in raw_text:
                score += 10

        if "realtor" in image_type:
            score -= 100

        return score

    def build_price_text(self, data):
        addition = data.get("articleAddition") or {}
        price = data.get("articlePrice") or {}
        detail = data.get("articleDetail") or {}

        trade_type = detail.get("tradeTypeName", "")
        deal_or_warrant = addition.get("dealOrWarrantPrc", "")
        rent_price = price.get("rentPrice", "")

        if trade_type == "월세" and deal_or_warrant and rent_price:
            return f"{deal_or_warrant}/{rent_price}"

        return deal_or_warrant

    def build_area_info(self, data):
        space = data.get("articleSpace") or {}
        detail = data.get("articleDetail") or {}

        supply = (
            space.get("supplySpace")
            or detail.get("area1")
            or ""
        )

        exclusive = (
            space.get("exclusiveSpace")
            or detail.get("area2")
            or ""
        )

        parts = []

        if supply:
            parts.append(f"공급 {supply}㎡")

        if exclusive:
            parts.append(f"전용 {exclusive}㎡")

        return " / ".join(parts)

    def build_article_title(self, data):
        detail = data.get("articleDetail") or {}

        print("=" * 80)
        print("[DETAIL KEYS]")
        print(list(detail.keys()))

        print("[COMPLEX NO]", detail.get("complexNo"))
        print("[HSCP NO]", detail.get("hscpNo"))

        print("[LATITUDE]", detail.get("latitude"))
        print("[LONGITUDE]", detail.get("longitude"))

        print("[CORTAR NO]", detail.get("cortarNo"))
        print("=" * 80)

        addition = data.get("articleAddition") or {}

        candidates = [
            detail.get("articleName", ""),
            addition.get("buildingName", ""),
            detail.get("buildingName", ""),
            detail.get("articleSubName", ""),
        ]

        cleaned = []

        for item in candidates:
            item = str(item or "").strip()

            if not item:
                continue

            if item in ["단독", "일반상가", "상가", "아파트"]:
                continue

            if item in cleaned:
                continue

            cleaned.append(item)

        title = " ".join(cleaned)
        title = " ".join(title.split())

        if not title:
            title = (
                detail.get("articleName")
                or detail.get("aptName")
                or addition.get("buildingName")
                or "네이버 부동산 매물"
            )

        return title[:255]

    def extract_images(self, data):
        result = []
        detail = data.get("articleDetail") or {}
        realtor = data.get("articleRealtor") or {}

        candidates = []

        article_photos = list(data.get("articlePhotos") or [])
        detail_article_photos = list(detail.get("articlePhotos") or [])
        building_photos = list(detail.get("buildingPhotos") or [])
        has_listing_photos = bool(
            article_photos
            or detail_article_photos
            or building_photos
        )
        real_estate_type = str(
            detail.get("realEstateTypeName")
            or detail.get("realEstateTypeCode")
            or ""
        ).strip().lower()
        non_residential_listing = any(
            keyword in real_estate_type
            for keyword in (
                "상가", "상업", "건물", "빌딩", "토지", "공장", "창고",
                "지식산업센터", "사무실", "commercial", "retail", "land",
                "factory", "warehouse",
            )
        )

        for img in article_photos:
            candidates.append(("article", "article_photo", img))

        for img in detail_article_photos:
            candidates.append(("article", "article_photo", img))

        # ---------------------------------------------------------
        # FLOOR PLAN FROM ARTICLE DATA
        # ---------------------------------------------------------

        for img in detail.get("grandPlanList") or []:
            candidates.append((
                "floorplan",
                "floorplan",
                img
            ))

        for img in detail.get("floorPlanList") or []:
            candidates.append((
                "floorplan",
                "floorplan",
                img
            ))

        for img in detail.get("articleFloorPlanList") or []:
            candidates.append((
                "floorplan",
                "floorplan",
                img
            ))

        floorplan_image = (
            detail.get("floorPlanPhoto")
            or detail.get("representativeFloorPlan")
            or ""
        )

        if floorplan_image:
            candidates.append((
                "floorplan",
                "floorplan",
                {
                    "imageSrc": floorplan_image
                }
            ))

        # ---------------------------------------------------------
        # FLOOR PLAN FROM RESPONSE LISTENER
        # ---------------------------------------------------------

        for img in self.extra_floorplan_images:
            candidates.append((
                "floorplan",
                "floorplan",
                img
            ))

        # ---------------------------------------------------------
        # GROUND PLAN GALLERY FROM rletNo
        # ---------------------------------------------------------
        # rletNo 기반 groundPlanGallery는 해당 매물의 단지 사진 후보로만 사용한다.
        # 실제 매물 사진이 하나도 없을 때 단지 공용사진을 대신 넣으면
        # 사진 없는 상가 등에 다른 아파트 사진이 노출될 수 있으므로 저장하지 않는다.
        if has_listing_photos and not non_residential_listing:
            for img in self.extra_ground_gallery_images:
                candidates.append((
                    "ground_gallery",
                    "ground_gallery",
                    img
                ))
        elif self.extra_ground_gallery_images:
            print(
                "[GROUND GALLERY SKIP]",
                detail.get("articleNo") or "",
                real_estate_type,
                "non_residential" if non_residential_listing else "no_listing_photo",
                len(self.extra_ground_gallery_images),
            )

        # ---------------------------------------------------------
        # IMAGES FROM ANY JSON RESPONSE LISTENER
        # ---------------------------------------------------------
        # 중요:
        # extra_response_images는 다른 단지/주변 API 이미지가 섞일 수 있어
        # 일반 매물 이미지로는 사용하지 않는다.
        # 안전한 floorplan 후보만 허용한다.
        for img in self.extra_response_images:
            category = img.get("imageCategory") or ""
            image_type = category

            if category != "floorplan":
                continue

            candidates.append((
                "floorplan",
                "floorplan",
                img
            ))

        # ---------------------------------------------------------
        # BUILDING
        # ---------------------------------------------------------

        for img in building_photos:
            candidates.append((
                "building",
                "building",
                img
            ))

        profile_image = (
            realtor.get("profileImageUrl")
            or realtor.get("profileFullImageUrl")
            or ""
        )

        if profile_image:
            candidates.append((
                "realtor",
                "realtor",
                {
                    "imageSrc": profile_image
                }
            ))

        temp = []
        seen = set()

        for category, image_type, img in candidates:
            image_url = (
                img.get("imageSrc")
                or img.get("imageUrl")
                or img.get("url")
                or ""
            )

            image_url = self.normalize_image_url(image_url)

            if not image_url:
                continue

            if self.is_bad_image_url(image_url):
                continue

            base_name = os.path.basename(
                image_url.split("?")[0]
            ).lower()

            image_key = (
                f"{category}|{base_name}"
            )

            if image_key in seen:
                continue

            seen.add(image_key)

            ext = "jpg"

            if ".png" in image_url.lower():
                ext = "png"
            elif ".webp" in image_url.lower():
                ext = "webp"

            score = self.image_priority_score(
                category,
                image_type,
                img,
            )

            temp.append({
                "score": score,
                "image_url": image_url,
                "local_path": "",
                "image_type": image_type,
                "image_category": category,
                "width": img.get("imageWidth") or img.get("width") or 0,
                "height": img.get("imageHeight") or img.get("height") or 0,
                "raw_json": json.dumps(img, ensure_ascii=False),
                "ext": ext,
            })

        temp.sort(
            key=lambda x: x.get("score", 0),
            reverse=True,
        )

        for idx, img in enumerate(temp):
            category = img.get("image_category") or "image"
            ext = img.get("ext") or "jpg"

            result.append({
                "image_order": idx + 1,
                "image_url": img.get("image_url", ""),
                "local_path": img.get("local_path", ""),
                "file_name": f"{category}_{idx + 1}.{ext}",
                "image_type": img.get("image_type", ""),
                "image_category": category,
                "width": img.get("width", 0),
                "height": img.get("height", 0),
                "raw_json": img.get("raw_json", ""),
            })

        return result


    def normalize_db_image_type(self, image_type, image_category):
        image_type = str(image_type or "").lower().strip()
        image_category = str(image_category or "").lower().strip()

        # blog_realestate_article_images.image_type enum 안전값:
        # main, building, floorplan, view, inside, outside, map, etc
        if image_category == "floorplan":
            return "floorplan"

        if image_category == "building":
            return "building"

        if image_category == "ground_gallery":
            return "building"

        if image_category == "article":
            return "inside"

        if image_category == "inside":
            return "inside"

        if image_category == "outside":
            return "outside"

        if image_category == "view":
            return "view"

        if image_category == "map":
            return "map"

        if image_category == "realtor":
            return "etc"

        if "floor" in image_type or "plan" in image_type:
            return "floorplan"

        if "building" in image_type:
            return "building"

        if "article" in image_type or "photo" in image_type:
            return "inside"

        return "etc"

    # ---------------------------------------------------------
    # IMAGE DOWNLOAD
    # ---------------------------------------------------------

    def safe_file_name(self, name):
        name = str(name or "").strip()

        if not name:
            return "image.jpg"

        bad_chars = [
            "\\",
            "/",
            ":",
            "*",
            "?",
            "\"",
            "<",
            ">",
            "|",
        ]

        for ch in bad_chars:
            name = name.replace(ch, "_")

        return name[:180]

    def get_image_storage_paths(
        self,
        article_no,
        category,
        file_name
    ):
        article_no = str(article_no or "").strip()
        category = str(category or "image").strip()
        file_name = self.safe_file_name(file_name)

        local_dir = os.path.join(
            REAL_ESTATE_STORAGE_DIR,
            article_no,
            category
        )

        os.makedirs(
            local_dir,
            exist_ok=True
        )

        file_path = os.path.join(
            local_dir,
            file_name
        )

        public_url = (
            f"{PUBLIC_STORAGE_BASE_URL}"
            f"/realestate/{article_no}/{category}/{urllib.parse.quote(file_name)}"
        )

        return file_path, public_url

    def download_image_to_local(
        self,
        image_url,
        article_no,
        category,
        file_name
    ):
        image_url = self.normalize_image_url(
            image_url
        )

        if not image_url:
            return {
                "success": False,
                "file_path": "",
                "public_url": "",
                "error": "empty image_url",
            }

        if self.is_bad_image_url(image_url):
            return {
                "success": False,
                "file_path": "",
                "public_url": "",
                "error": "bad image url filtered",
            }

        file_path, public_url = self.get_image_storage_paths(
            article_no,
            category,
            file_name
        )

        if os.path.exists(file_path) and os.path.getsize(file_path) > 0:
            return {
                "success": True,
                "file_path": file_path,
                "public_url": public_url,
                "error": "",
            }

        try:
            req = urllib.request.Request(
                image_url,
                headers={
                    "User-Agent": USER_AGENT,
                    "Referer": "https://new.land.naver.com/",
                    "Accept": "image/avif,image/webp,image/apng,image/svg+xml,image/*,*/*;q=0.8",
                }
            )

            with urllib.request.urlopen(
                req,
                timeout=15
            ) as response:
                content = response.read()

            if not content:
                return {
                    "success": False,
                    "file_path": "",
                    "public_url": "",
                    "error": "empty content",
                }

            with open(
                file_path,
                "wb"
            ) as f:
                f.write(content)

            return {
                "success": True,
                "file_path": file_path,
                "public_url": public_url,
                "error": "",
            }

        except Exception as e:
            return {
                "success": False,
                "file_path": "",
                "public_url": "",
                "error": str(e),
            }


    # ---------------------------------------------------------
    # SAVE
    # ---------------------------------------------------------

    def save_article_detail(self, article_id, data):
        detail = data.get("articleDetail") or {}
        addition = data.get("articleAddition") or {}

        row = {
            "article_name": self.build_article_title(data),
            "trade_type": detail.get("tradeTypeName", ""),
            "real_estate_type": detail.get("realEstateTypeName", ""),
            "price_text": self.build_price_text(data),
            "building_name": (
                addition.get("buildingName", "")
                or detail.get("buildingName", "")
            ),
            "floor_info": addition.get("floorInfo", ""),
            "area_info": self.build_area_info(data),
            "article_feature_desc": (
                detail.get("articleFeatureDesc")
                or detail.get("detailDescription")
                or detail.get("articleDescription")
                or ""
            ),
            "direction_code": (
                detail.get("direction")
                or addition.get("direction")
                or ""
            ),
            "raw_json": json.dumps(
                data,
                ensure_ascii=False
            ),
        }

        self.update_dynamic_by_id(
            "blog_realtor_articles",
            article_id,
            row,
        )

    def save_article_images(self, article_no, images):
        if not images:
            print("[IMAGE SAVE] 0")
            return

        columns = self.get_columns("blog_realestate_article_images")

        conn = get_conn()

        try:
            saved = 0
            representative_set = False

            with conn.cursor() as cur:
                # 재수집 시 이전에 잘못 저장된 단지/갤러리 이미지를 확실히 제거한다.
                # image_category가 NULL인 과거 데이터도 삭제 대상에 포함한다.
                cur.execute(
                    """
                    DELETE FROM blog_realestate_article_images
                    WHERE article_no = %s
                      AND (
                            image_category IS NULL
                            OR image_category <> 'header_image'
                          )
                    """,
                    (article_no,)
                )

                print("[IMAGE OLD DELETE]", article_no, cur.rowcount)

                for img in images:
                    image_url = img.get("image_url", "")
                    image_category = img.get("image_category", "") or "image"
                    file_name = img.get("file_name", "")

                    db_image_type = self.normalize_db_image_type(
                        img.get("image_type", ""),
                        image_category,
                    )

                    # 대표이미지 기준:
                    # 평면도/중개사 이미지는 제외하고, 실제 매물/건물 이미지 중 첫 번째를 대표로 지정합니다.
                    is_representative = 0

                    if not representative_set:
                        if image_category not in ["floorplan", "realtor"]:
                            if db_image_type in ["inside", "building", "outside", "view"]:
                                is_representative = 1
                                representative_set = True

                    download_result = self.download_image_to_local(
                        image_url=image_url,
                        article_no=article_no,
                        category=image_category,
                        file_name=file_name,
                    )

                    if download_result.get("success"):
                        local_path = download_result.get("public_url", "")
                        local_file_path = download_result.get("file_path", "")
                        download_status = "downloaded"

                        print(
                            "[IMAGE DOWNLOADED]",
                            article_no,
                            image_category,
                            file_name,
                        )

                    else:
                        local_path = image_url
                        local_file_path = ""
                        download_status = "failed"

                        print(
                            "[IMAGE DOWNLOAD FAIL]",
                            article_no,
                            image_category,
                            file_name,
                            download_result.get("error", ""),
                        )

                    raw_json_text = img.get("raw_json", "")

                    try:
                        raw_obj = json.loads(raw_json_text) if raw_json_text else {}
                    except Exception:
                        raw_obj = {
                            "raw": raw_json_text
                        }

                    raw_obj["_download"] = {
                        "success": bool(download_result.get("success")),
                        "file_path": local_file_path,
                        "public_url": local_path,
                        "error": download_result.get("error", ""),
                    }

                    data = {
                        "article_no": article_no,
                        "image_url": image_url,
                        "local_path": local_path,
                        "local_file_path": local_file_path,
                        "public_url": local_path,
                        "file_name": file_name,
                        "image_type": db_image_type,
                        "image_category": image_category,
                        "image_source": "naver_land",
                        "download_status": download_status,
                        "sort_order": img.get("image_order", 0),
                        "is_representative": is_representative,
                        "width": img.get("width", 0),
                        "height": img.get("height", 0),
                        "raw_json": json.dumps(raw_obj, ensure_ascii=False),
                    }

                    filtered = {
                        key: value
                        for key, value in data.items()
                        if key in columns
                    }

                    if not filtered:
                        continue

                    keys = list(filtered.keys())
                    placeholders = ", ".join(["%s"] * len(keys))
                    column_sql = ", ".join(keys)
                    values = [filtered[key] for key in keys]

                    update_sql = ", ".join([
                        f"{key}=VALUES({key})"
                        for key in keys
                        if key not in ["id", "created_at"]
                    ])

                    sql = f"""
                    INSERT INTO blog_realestate_article_images
                    ({column_sql})
                    VALUES ({placeholders})
                    ON DUPLICATE KEY UPDATE
                    {update_sql}
                    """

                    cur.execute(sql, values)
                    saved += 1

            conn.commit()
            print("[IMAGE SAVE]", saved)

        finally:
            conn.close()

    def save_realtor(self, article_no, data):
        realtor = data.get("articleRealtor") or {}

        if not realtor:
            print("[REALTOR SAVE] empty")
            return

        profile_image_url = self.normalize_image_url(
            realtor.get("profileImageUrl")
            or realtor.get("profileFullImageUrl")
            or ""
        )

        row = {
            "article_no": article_no,
            "office_name": realtor.get("realtorName", ""),
            "representative_name": realtor.get("representativeName", ""),
            "realtor_name": realtor.get("realtorName", ""),
            "phone": realtor.get("representativeTelNo", ""),
            "mobile": realtor.get("cellPhoneNo", ""),
            "address": realtor.get("address", ""),
            "profile_image_url": profile_image_url,
            "raw_json": json.dumps(realtor, ensure_ascii=False),
        }

        self.insert_dynamic(
            "blog_realestate_article_realtors",
            row,
            unique_update=True,
        )

        print("[REALTOR SAVE]", row.get("office_name"))

    def clean_school_name(self, name, school_type):
        name = str(name or "").strip()
        school_type = str(school_type or "").strip()

        if not name:
            return ""

        name = re.sub(r"[\s,./·ㆍ:;!~()\[\]{}]+$", "", name)
        name = name.replace("초교", "초등학교")
        name = name.replace("중교", "중학교")
        name = name.replace("고교", "고등학교")

        generic_words = [
            "초등학교", "중학교", "고등학교", "학교",
            "인근초등학교", "인근중학교", "인근고등학교",
            "초등학교근접", "학교근접", "학교인근",
        ]

        if name in generic_words:
            return ""

        # -----------------------------------------------------
        # STEP357: V2 학교명 오인식 차단 검증 반영
        # V2와 동일한 필터를 유지하며 수집/저장/대상선정 로직은 변경하지 않는다.
        # 학교명 오인식 차단
        # 예: 보유중학교, 공인중학교, 흐르고등학교, 생각하고등학교 등
        # -----------------------------------------------------
        raw_name = name
        base = raw_name

        for suffix in ["초등학교", "중학교", "고등학교", "초", "중", "고"]:
            if base.endswith(suffix):
                base = base[: -len(suffix)]
                break

        base = base.strip()

        bad_exact = {
            "보유", "공인", "중개", "부동산", "대표", "상담", "문의", "연락",
            "매물", "전문", "최다", "책임", "전화", "급매", "추천", "강추",
            "생각", "생각하", "흐르", "수리되", "조용하", "조용", "가깝", "가까", "좋", "넓",
            "크", "작", "많", "적", "높", "낮", "있", "없", "가능", "근접",
            "인접", "주변", "인근", "도보", "바로", "구조", "채광", "통풍",
        }

        bad_contains = [
            "공인", "중개", "부동산", "대표", "상담", "문의", "연락",
            "보유", "최다", "전문", "매물", "전화", "번호", "오피스텔",
            "아파트", "상가", "단지", "로얄", "타입", "뷰", "월",
        ]

        bad_endings = [
            "하고", "되고", "있고", "없고", "같고", "좋고", "넓고", "크고",
            "작고", "많고", "적고", "높고", "낮고", "흐르고", "생각하고",
        ]

        if not base or len(base) < 2 or len(base) > 10:
            return ""

        if base in bad_exact:
            return ""

        if any(x in base for x in bad_contains):
            return ""

        if any(base.endswith(x) for x in bad_endings):
            return ""

        if re.search(r"\d", base):
            return ""

        if school_type == "초등학교":
            if name.endswith("초"):
                name = name[:-1] + "초등학교"
            elif not name.endswith("초등학교"):
                name = name + "초등학교"

        elif school_type == "중학교":
            if name.endswith("중"):
                name = name[:-1] + "중학교"
            elif not name.endswith("중학교"):
                name = name + "중학교"

        elif school_type == "고등학교":
            if name.endswith("고"):
                name = name[:-1] + "고등학교"
            elif name.endswith("고교"):
                name = name[:-2] + "고등학교"
            elif not name.endswith("고등학교"):
                name = name + "고등학교"

        name = name.replace("초등학교등학교", "초등학교")
        name = name.replace("중학교학교", "중학교")
        name = name.replace("고등학교등학교", "고등학교")

        if len(name) < 4:
            return ""

        return name

    def add_school_candidate(self, schools, seen, name, school_type, level, source, distance_text="매물 설명 기준 인근"):
        school_name = self.clean_school_name(name, school_type)

        if not school_name:
            return

        key = f"{school_name}|{school_type}"

        if key in seen:
            return

        seen.add(key)

        schools.append({
            "school_name": school_name,
            "school_type": school_type,
            "school_level": level,
            "distance_text": distance_text,
            "distance_meter": 0,
            "raw_json": json.dumps({
                "source": source,
                "original_name": name,
            }, ensure_ascii=False)
        })

    def extract_schools_from_description(self, data):
        detail = data.get("articleDetail") or {}
        addition = data.get("articleAddition") or {}

        text_parts = [
            detail.get("detailDescription"),
            detail.get("articleFeatureDescription"),
            detail.get("articleFeatureDesc"),
            addition.get("articleFeatureDesc"),
            addition.get("articleFeatureDescription"),
        ]

        text = "\n".join([
            str(x or "")
            for x in text_parts
            if str(x or "").strip()
        ])

        schools = []
        seen = set()

        # -----------------------------------------------------
        # 1) 정식 학교명: 해밀초등학교 / 해밀중학교 / 해밀고등학교
        # -----------------------------------------------------
        patterns = [
            (r"([가-힣A-Za-z0-9]{2,30}초등학교)", "초등학교", "elementary"),
            (r"([가-힣A-Za-z0-9]{2,30}중학교)", "중학교", "middle"),
            (r"([가-힣A-Za-z0-9]{2,30}고등학교)", "고등학교", "high"),
        ]

        for pattern, school_type, level in patterns:
            for name in re.findall(pattern, text):
                self.add_school_candidate(
                    schools,
                    seen,
                    name,
                    school_type,
                    level,
                    "description_full_name",
                )

        # -----------------------------------------------------
        # 2) 축약형: 해밀초 / 해밀중 / 해밀고
        # -----------------------------------------------------
        short_patterns = [
            (r"(?:^|[^가-힣A-Za-z0-9])([가-힣A-Za-z0-9]{2,8})초(?!등학교)", "초등학교", "elementary", "초"),
            (r"(?:^|[^가-힣A-Za-z0-9])([가-힣A-Za-z0-9]{2,8})중(?!학교)", "중학교", "middle", "중"),
            (r"(?:^|[^가-힣A-Za-z0-9])([가-힣A-Za-z0-9]{2,8})고(?:교)?(?!등학교)", "고등학교", "high", "고"),
        ]

        for pattern, school_type, level, suffix in short_patterns:
            for prefix in re.findall(pattern, text):
                prefix = str(prefix or "").strip()

                # 조사/문장 일부가 붙는 오탐 방지
                prefix = re.sub(r".*[\s\n\r]", "", prefix)
                prefix = re.sub(r"^[,./·ㆍ]+", "", prefix)

                if prefix in ["초등학교", "중학교", "고등학교", "학교", "근접", "인근"]:
                    continue

                # 오탐 방지: 가깝고/좋고/넓고 같은 형용사 어간을 학교명으로 인식하지 않도록 차단
                if len(prefix) < 2:
                    continue

                school_stop_prefixes = [
                    "가깝", "좋", "넓", "크", "작", "많", "적",
                    "밝", "낮", "높", "빠르", "편하", "깨끗",
                    "가능", "근접", "인접", "도보", "주변", "인근",
                ]

                if any(prefix.endswith(x) or prefix == x for x in school_stop_prefixes):
                    continue

                self.add_school_candidate(
                    schools,
                    seen,
                    prefix + suffix,
                    school_type,
                    level,
                    "description_short_name",
                )

        # -----------------------------------------------------
        # 3) 복합 축약형: 해밀초,중,고교 / 해밀 초중고 / 해밀초중고
        # -----------------------------------------------------
        combo_patterns = [
            r"(?:^|[^가-힣A-Za-z0-9])([가-힣A-Za-z0-9]{2,8})초\s*[,·ㆍ/]?\s*중\s*[,·ㆍ/]?\s*고(?:교|등학교)?",
            r"(?:^|[^가-힣A-Za-z0-9])([가-힣A-Za-z0-9]{2,8})\s*초\s*중\s*고(?:교|등학교)?",
        ]

        for pattern in combo_patterns:
            for prefix in re.findall(pattern, text):
                prefix = str(prefix or "").strip()
                prefix = re.sub(r".*[\s\n\r]", "", prefix)

                if not prefix:
                    continue

                self.add_school_candidate(schools, seen, prefix + "초", "초등학교", "elementary", "description_combo")
                self.add_school_candidate(schools, seen, prefix + "중", "중학교", "middle", "description_combo")
                self.add_school_candidate(schools, seen, prefix + "고", "고등학교", "high", "description_combo")

        print("[SCHOOL EXTRACT]", len(schools))

        for school in schools:
            print(
                "[SCHOOL FOUND]",
                school.get("school_name"),
                school.get("school_type"),
            )

        return schools

    def save_schools(self, article_no, schools):
        # 최종 DB 저장 직전에도 설명문 기반 학교명을 다시 검증한다.
        # 이전 병합 경로가 남아 있어도 "가까고등학교" 같은 문장 오탐은 저장하지 않는다.
        validated_schools = []
        for school in schools or []:
            item = dict(school)
            raw_json = str(item.get("raw_json") or "")
            if '"source": "description_' in raw_json:
                cleaned_name = self.clean_school_name(
                    item.get("school_name"),
                    item.get("school_type"),
                )
                if not cleaned_name:
                    print("[SCHOOL REJECT]", item.get("school_name"), "description_false_positive")
                    continue
                item["school_name"] = cleaned_name
            validated_schools.append(item)
        schools = validated_schools

        if not schools:
            print("[SCHOOL SAVE] 0")
            return

        columns = self.get_columns(
            "blog_realestate_article_schools"
        )

        conn = get_conn()

        try:
            with conn.cursor() as cur:

                cur.execute(
                    """
                    DELETE FROM blog_realestate_article_schools
                    WHERE article_no = %s
                    """,
                    (article_no,)
                )

                sort_order = 1

                for school in schools:

                    row = {
                        "article_no": article_no,
                        "complex_number": school.get("complex_number", ""),
                        "school_name": school["school_name"],
                        "school_type": school["school_type"],
                        "school_level": school["school_level"],
                        "sort_order": sort_order,
                        "distance_text": school["distance_text"],
                        "distance_meter": school["distance_meter"],
                        "raw_json": school["raw_json"],
                    }

                    filtered = {
                        k: v
                        for k, v in row.items()
                        if k in columns
                    }

                    keys = list(filtered.keys())

                    sql = f"""
                    INSERT INTO
                    blog_realestate_article_schools
                    ({",".join(keys)})
                    VALUES
                    ({",".join(["%s"] * len(keys))})
                    """

                    cur.execute(
                        sql,
                        [filtered[k] for k in keys]
                    )

                    sort_order += 1

            conn.commit()

            print(
                "[SCHOOL SAVE]",
                len(schools)
            )

        finally:
            conn.close()

    # ---------------------------------------------------------
    # FACILITY / LIFESTYLE
    # ---------------------------------------------------------

    def extract_facility_text(self, data):
        detail = data.get("articleDetail") or {}
        addition = data.get("articleAddition") or {}

        text_parts = [
            detail.get("detailDescription"),
            detail.get("articleFeatureDescription"),
            detail.get("articleFeatureDesc"),
            addition.get("articleFeatureDesc"),
            addition.get("articleFeatureDescription"),
        ]

        return "\n".join([
            str(x or "")
            for x in text_parts
            if str(x or "").strip()
        ])

    def normalize_facility_name(self, name):
        name = str(name or "").strip()
        name = name.replace("\r", " ").replace("\n", " ")
        name = " ".join(name.split())
        name = re.sub(r"^[,./·ㆍ\-\s]+", "", name)
        name = re.sub(r"[,./·ㆍ\-\s]+$", "", name)
        return name[:80]

    def add_facility_candidate(self, facilities, seen, name, facility_type, category, source, distance_text="매물 설명 기준 인근"):
        facility_name = self.normalize_facility_name(name)

        if not facility_name:
            return

        bad_names = [
            "있습니다", "가능합니다", "좋습니다", "가깝습니다", "도보 가능합니다",
            "주변", "인근", "근처", "생활권", "편의시설",
        ]

        if facility_name in bad_names:
            return

        if len(facility_name) < 2:
            return

        key = f"{facility_type}|{facility_name}"

        if key in seen:
            return

        seen.add(key)

        facilities.append({
            "facility_name": facility_name,
            "facility_type": facility_type,
            "facility_category": category,
            "distance_text": distance_text,
            "distance_meter": 0,
            "raw_json": json.dumps({
                "source": source,
                "original_name": name,
            }, ensure_ascii=False),
        })

    def extract_facilities_from_description(self, data):
        text = self.extract_facility_text(data)

        facilities = []
        seen = set()

        if not text:
            print("[FACILITY EXTRACT] 0")
            return facilities

        # -----------------------------------------------------
        # 1) 명확한 생활 키워드
        # -----------------------------------------------------
        keyword_rules = [
            ("BRT", "교통", "transport", "BRT 정류장"),
            ("정류장", "교통", "transport", "정류장"),
            ("버스", "교통", "transport", "버스 이용권"),
            ("지하철", "교통", "transport", "지하철역"),
            ("역", "교통", "transport", "역세권"),
            ("공원", "공원/녹지", "park", "공원"),
            ("녹지", "공원/녹지", "park", "녹지"),
            ("병원", "의료", "medical", "병원"),
            ("의원", "의료", "medical", "의원"),
            ("약국", "의료", "medical", "약국"),
            ("마트", "쇼핑/편의", "shopping", "마트"),
            ("백화점", "쇼핑/편의", "shopping", "백화점"),
            ("시장", "쇼핑/편의", "shopping", "시장"),
            ("편의점", "쇼핑/편의", "shopping", "편의점"),
            ("도서관", "공공/문화", "public", "도서관"),
            ("주민센터", "공공/문화", "public", "주민센터"),
            ("행정복지센터", "공공/문화", "public", "행정복지센터"),
            ("수영장", "커뮤니티", "community", "수영장"),
            ("사우나", "커뮤니티", "community", "사우나"),
            ("헬스", "커뮤니티", "community", "헬스장"),
            ("골프", "커뮤니티", "community", "골프연습장"),
            ("게스트하우스", "커뮤니티", "community", "게스트하우스"),
        ]

        for keyword, facility_type, category, display_name in keyword_rules:
            if keyword in text:
                self.add_facility_candidate(
                    facilities,
                    seen,
                    display_name,
                    facility_type,
                    category,
                    "description_keyword",
                )

        # -----------------------------------------------------
        # 2) 고유명사형 시설 추출: 기쁨뜰공원, 충남대학교 병원 등
        # -----------------------------------------------------
        entity_patterns = [
            (r"([가-힣A-Za-z0-9]{2,30}\s?공원)", "공원/녹지", "park"),
            (r"([가-힣A-Za-z0-9]{2,30}\s?병원)", "의료", "medical"),
            (r"([가-힣A-Za-z0-9]{2,30}\s?마트)", "쇼핑/편의", "shopping"),
            (r"([가-힣A-Za-z0-9]{2,30}\s?백화점)", "쇼핑/편의", "shopping"),
            (r"([가-힣A-Za-z0-9]{2,30}\s?시장)", "쇼핑/편의", "shopping"),
            (r"([가-힣A-Za-z0-9]{2,30}\s?도서관)", "공공/문화", "public"),
            (r"([가-힣A-Za-z0-9]{2,30}\s?정류장)", "교통", "transport"),
        ]

        for pattern, facility_type, category in entity_patterns:
            for name in re.findall(pattern, text):
                name = self.normalize_facility_name(name)

                # 문장 일부가 과하게 붙는 경우 마지막 공백 뒤 단어만 우선 사용
                if " " in name and len(name) > 14:
                    parts = name.split()
                    name = " ".join(parts[-2:]) if len(parts[-1]) <= 4 else parts[-1]

                self.add_facility_candidate(
                    facilities,
                    seen,
                    name,
                    facility_type,
                    category,
                    "description_entity",
                )

        print("[FACILITY EXTRACT]", len(facilities))

        for facility in facilities:
            print(
                "[FACILITY FOUND]",
                facility.get("facility_type"),
                facility.get("facility_name"),
            )

        return facilities

    def save_facilities(self, article_no, facilities):
        table_name = "blog_realestate_article_facilities"

        if not facilities:
            print("[FACILITY SAVE] 0")
            return

        if not self.table_exists(table_name):
            print("[FACILITY SAVE SKIP] table not found:", table_name)
            print("[FACILITY SAVE SKIP] 먼저 migration_blog_realestate_article_facilities_0530.sql 실행 필요")
            return

        columns = self.get_columns(table_name)

        conn = get_conn()

        try:
            with conn.cursor() as cur:
                cur.execute(
                    f"""
                    DELETE FROM {table_name}
                    WHERE article_no = %s
                    """,
                    (article_no,)
                )

                sort_order = 1

                for facility in facilities:
                    row = {
                        "article_no": article_no,
                        "facility_name": facility.get("facility_name", ""),
                        "facility_type": facility.get("facility_type", ""),
                        "facility_category": facility.get("facility_category", ""),
                        "sort_order": sort_order,
                        "distance_text": facility.get("distance_text", ""),
                        "distance_meter": facility.get("distance_meter", 0),
                        "raw_json": facility.get("raw_json", ""),
                    }

                    filtered = {
                        k: v
                        for k, v in row.items()
                        if k in columns
                    }

                    if not filtered:
                        continue

                    keys = list(filtered.keys())

                    sql = f"""
                    INSERT INTO {table_name}
                    ({','.join(keys)})
                    VALUES
                    ({','.join(['%s'] * len(keys))})
                    """

                    cur.execute(
                        sql,
                        [filtered[k] for k in keys]
                    )

                    sort_order += 1

            conn.commit()
            print("[FACILITY SAVE]", len(facilities))

        finally:
            conn.close()

    # ---------------------------------------------------------
    # BROWSER
    # ---------------------------------------------------------

    def start(self):
        print("=" * 80)
        print("[BROWSER START]")

        self.playwright = sync_playwright().start()

        self.browser = self.playwright.chromium.launch(
            headless=self.headless,
            args=[
                "--disable-blink-features=AutomationControlled",
                "--no-sandbox",
            ],
        )

        self.context = self.browser.new_context(
            locale="ko-KR",
            viewport={
                "width": 1365,
                "height": 900,
            },
            user_agent=USER_AGENT,
        )

        self.page = self.context.new_page()
        self.page.on("response", self.handle_response)

        self.page.goto(
            "https://new.land.naver.com/",
            wait_until="commit",
            timeout=30000,
        )

        print("[NAVER READY]")

    def close(self):
        print("[BROWSER CLOSED]")

        try:
            if self.browser:
                self.browser.close()
        except Exception:
            pass

        try:
            if self.playwright:
                self.playwright.stop()
        except Exception:
            pass

    # ---------------------------------------------------------
    # FETCH
    # ---------------------------------------------------------

    def build_url(self, article_no):
        return (
            "https://new.land.naver.com/complexes"
            f"?articleNo={article_no}"
        )

    def is_target_response(self, response, article_no):
        url = str(response.url).lower()

        return (
            "new.land.naver.com/api/articles/" in url
            and str(article_no) in url
        )

    def fetch_one(self, row):
        article_id = row["id"]
        article_no = row["article_no"]
        realtor_id = row.get("realtor_id")

        self.extra_floorplan_images = []
        self.extra_response_images = []

        print("-" * 80)
        print("[REALTOR]", realtor_id)
        print("[ARTICLE]", article_no)
        started_at = time.time()

        url = self.build_url(article_no)

        print("[OPEN]", url)

        try:
            with self.page.expect_response(
                lambda r: self.is_target_response(r, article_no),
                timeout=10000,
            ) as response_info:
                self.page.goto(
                    url,
                    wait_until="commit",
                    timeout=30000,
                )

            response = response_info.value

            print("[ARTICLE RESPONSE]", response.status)

            if response.status != 200:
                self.mark_failed(article_id, f"status={response.status}")
                return False

            data = response.json()

            # HTTP 200이어도 종료/비정상 매물은 articleDetail이 빈 JSON으로
            # 내려올 수 있다. 이를 정상 저장하면 기존 제목·가격·raw_json이
            # placeholder로 덮이므로, 기존 정상 상세정보를 그대로 보존한다.
            response_detail = data.get("articleDetail") if isinstance(data, dict) else None
            if not isinstance(response_detail, dict) or not response_detail:
                print(
                    "[ARTICLE DETAIL EMPTY - EXISTING DATA PRESERVED]",
                    article_no,
                )
                return False

            # 법정 표시광고 소재지 보강:
            # 매물 API가 detailAddress를 숨기는 경우 hscpNo 기반 단지 주소를 raw_json에 병합한다.
            data = self.fetch_complex_address_detail(data)

            self.trigger_lazy_image_requests()
            self.fetch_floorplan_extra_apis(data)
            self.fetch_ground_plan_gallery(data)

            # 네이버 통합검색 보강:
            # 공식 아파트 단지명만 검색하고, 동일 단지의 네이버페이 부동산 결과가
            # 확인된 경우에만 배정학교/단지정보/AI 브리핑 참고 근거를 저장한다.
            search_enrichment = collect_naver_search_enrichment(
                self.context,
                data,
            )
            data["naverSearchEnrichment"] = search_enrichment
            print(
                "[NAVER SEARCH ENRICHMENT]",
                "eligible=", search_enrichment.get("eligible"),
                "verified=", search_enrichment.get("verified"),
                "query=", search_enrichment.get("query") or "-",
                "reason=", search_enrichment.get("reason"),
            )

            # STEP25 FAST:
            # school/json path 탐색 로그는 운영 수집에서는 제외한다.
            time.sleep(0.5)

            detail = data.get("articleDetail") or {}
            article_name = detail.get("articleName") or ""

            print("[CAPTURED]", article_no, article_name)
            print("[EXTRA FLOORPLAN]", len(self.extra_floorplan_images))
            print("[EXTRA RESPONSE IMAGES]", len(self.extra_response_images))

            self.save_article_detail(article_id, data)

            images = self.extract_images(data)

            image_category_summary = {}
            for img in images:
                category = img.get("image_category") or "unknown"
                image_category_summary[category] = image_category_summary.get(category, 0) + 1

            print("[IMAGE CATEGORY SUMMARY]", image_category_summary)

            self.save_article_images(article_no, images)

            self.save_realtor(article_no, data)

            if search_enrichment.get("verified"):
                assigned_schools = search_enrichment.get("assigned_schools") or []
                # 동일 단지의 네이버페이 배정학교가 있으면 설명문 추출값을 사용하지 않는다.
                # 배정학교가 없을 때만 정식 학교명(축약형 제외)을 보조값으로 제한한다.
                schools = []
                existing = set()
                for item in assigned_schools:
                    key = re.sub(
                        r"[\s\-_·ㆍ.,()\[\]]+",
                        "",
                        str(item.get("school_name") or ""),
                    ).lower()
                    if not key or key in existing:
                        continue
                    existing.add(key)
                    schools.append({
                        "complex_number": "",
                        "school_name": item.get("school_name") or "",
                        "school_type": item.get("school_type") or "",
                        "school_level": item.get("school_level") or "",
                        "distance_text": "",
                        "distance_meter": 0,
                        "raw_json": json.dumps({
                            "source": "naver_pay_realestate",
                            "assignment_status": "assigned",
                            "search_query": search_enrichment.get("query") or "",
                            "collected_at": search_enrichment.get("collected_at") or "",
                        }, ensure_ascii=False),
                    })
                if not assigned_schools:
                    description_schools = self.extract_schools_from_description(data)
                    schools = [
                        item for item in description_schools
                        if '"source": "description_full_name"' in (item.get("raw_json") or "")
                    ]
                print("[ASSIGNED SCHOOL MERGE]", len(assigned_schools), "total=", len(schools))
            else:
                schools = self.extract_schools_from_description(data)

            self.save_schools(
                article_no,
                schools
            )

            if search_enrichment.get("verified"):
                nearby_facilities = search_enrichment.get("nearby_facilities") or []
                # 네이버페이/AI 브리핑의 확인된 주변정보를 먼저 두고,
                # 기존 매물 설명의 시설 정보는 중복되지 않는 보조값으로만 뒤에 붙인다.
                facilities = []
                existing = set()
                for item in nearby_facilities:
                    key = re.sub(
                        r"[\s\-_·ㆍ.,()\[\]]+",
                        "",
                        str(item.get("facility_name") or ""),
                    ).lower()
                    if not key or key in existing:
                        continue
                    existing.add(key)
                    facilities.append({
                        "facility_name": item.get("facility_name") or "",
                        "facility_type": item.get("facility_type") or "",
                        "facility_category": item.get("facility_category") or "",
                        "distance_text": "",
                        "distance_meter": 0,
                        "raw_json": json.dumps({
                            "source": "naver_pay_realestate",
                            "search_query": search_enrichment.get("query") or "",
                            "collected_at": search_enrichment.get("collected_at") or "",
                        }, ensure_ascii=False),
                    })
                for item in self.extract_facilities_from_description(data):
                    key = re.sub(
                        r"[\s\-_·ㆍ.,()\[\]]+",
                        "",
                        str(item.get("facility_name") or ""),
                    ).lower()
                    if not key or key in existing:
                        continue
                    existing.add(key)
                    facilities.append(item)
                print("[SEARCH FACILITY MERGE]", len(nearby_facilities), "total=", len(facilities))
            else:
                facilities = self.extract_facilities_from_description(data)

            self.save_facilities(
                article_no,
                facilities
            )

            self.mark_success(article_id)

            elapsed = time.time() - started_at
            print("[FETCH ONE DONE]", article_no, f"{elapsed:.1f}s")

            return True

        except Exception as e:
            print("[FETCH ERROR]", str(e))
            self.mark_failed(article_id, str(e))
            return False

    # ---------------------------------------------------------
    # RUN
    # ---------------------------------------------------------

    def run(self, limit=20, realtor_id=None, article_no=None):
        targets = self.load_targets(
            limit=limit,
            realtor_id=realtor_id,
            article_no=article_no,
        )

        print("=" * 80)
        print("[TARGET COUNT]", len(targets), "realtor_id=", realtor_id)

        if not targets:
            return

        self.start()

        try:
            success = 0
            failed = 0

            for row in targets:
                ok = self.fetch_one(row)

                if ok:
                    success += 1
                else:
                    failed += 1

                time.sleep(0.3)

            print("=" * 80)
            print("[DONE]")
            print("[SUCCESS]", success)
            print("[FAILED]", failed)

        finally:
            self.close()



def main():
    import argparse

    parser = argparse.ArgumentParser()
    parser.add_argument("--limit", type=int, default=20)
    parser.add_argument("--realtor-id", type=int, default=None)
    parser.add_argument("--article-no", type=str, default=None)
    parser.add_argument("--headless", type=int, default=1)
    parser.add_argument("--lock-ttl-minutes", type=int, default=240)
    parser.add_argument("--no-db-lock", action="store_true")

    args = parser.parse_args()

    lock_name = (
        f"naver_response_article_fetcher_{args.realtor_id}"
        if args.realtor_id
        else "naver_response_article_fetcher"
    )
    lock_owner = None
    run_id = None
    final_status = "success"
    final_message = "naver_response_article_fetcher completed"

    if not args.no_db_lock:
        locked, lock_owner = acquire_db_lock(
            lock_name,
            ttl_minutes=args.lock_ttl_minutes,
        )

        if not locked:
            print("[SKIP] naver_response_article_fetcher already running")
            return

    fetcher = NaverResponseArticleFetcher(
        headless=bool(args.headless)
    )

    try:
        run_id = create_pipeline_run(
            "fetch_article_detail",
            message="naver_response_article_fetcher started",
            meta={
                "limit": args.limit,
                "headless": args.headless,
                "realtor_id": args.realtor_id,
            },
        )

        fetcher.run(
            limit=args.limit,
            realtor_id=args.realtor_id,
            article_no=args.article_no,
        )

    except Exception as e:
        final_status = "failed"
        final_message = str(e)[:2000]
        print("[FATAL]", final_message)
        traceback.print_exc()
        add_pipeline_log(
            run_id=run_id,
            level="error",
            step_name="fatal",
            message=final_message,
            context={"traceback": traceback.format_exc()},
        )

    finally:
        finish_pipeline_run(
            run_id,
            final_status,
            message=final_message,
            meta={
                "limit": args.limit,
                "headless": args.headless,
                "realtor_id": args.realtor_id,
            },
        )

        if lock_owner:
            release_db_lock(lock_name, lock_owner)


if __name__ == "__main__":
    main()
