#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""빠른 웹진 전용 네이버 현재 매물번호 수집기."""

from __future__ import annotations

import argparse
import json
import re
from typing import Any

from playwright.sync_api import sync_playwright


FRONT_API = "https://fin.land.naver.com/front-api/v1/realtor/articles"


def uniq(items: list[str]) -> list[str]:
    seen: set[str] = set()
    output: list[str] = []
    for value in items:
        text = str(value or "").strip()
        if text and text not in seen:
            seen.add(text)
            output.append(text)
    return output


def extract_article_nos(text: str) -> list[str]:
    patterns = (
        r'"articleNo"\s*:\s*"?([0-9]{8,})"?',
        r'"atclNo"\s*:\s*"?([0-9]{8,})"?',
        r'"articleNumber"\s*:\s*"?([0-9]{8,})"?',
    )
    found: list[str] = []
    for pattern in patterns:
        found.extend(re.findall(pattern, str(text or ""), flags=re.I))
    return uniq(found)


def normalize_trade(value: Any) -> str:
    text = str(value or "").strip().upper()
    if text in {"A1", "매매"}:
        return "A1"
    if text in {"B1", "전세"}:
        return "B1"
    if text in {"B2", "B3", "월세", "단기임대"}:
        return "B2"
    return ""


def extract_article_map(payload: Any) -> dict[str, dict[str, str]]:
    """목록 응답에서 거래유형과 네이버 확인일을 매물번호별로 추출한다."""
    result: dict[str, dict[str, str]] = {}

    def visit(value: Any) -> None:
        if isinstance(value, dict):
            article_no = str(value.get("articleNo") or value.get("atclNo") or value.get("articleNumber") or "").strip()
            trade = normalize_trade(value.get("tradeTypeCode") or value.get("tradeType") or value.get("tradeTypeName"))
            registered_at = str(
                value.get("articleConfirmYMD")
                or value.get("articleConfirmYmd")
                or value.get("confirmYmd")
                or value.get("exposeStartYMD")
                or value.get("exposeStartYmd")
                or ""
            ).strip()
            if article_no:
                item = result.setdefault(article_no, {})
                if trade:
                    item["trade_type"] = trade
                if registered_at:
                    item["registered_at"] = registered_at
            for child in value.values():
                visit(child)
        elif isinstance(value, list):
            for child in value:
                visit(child)

    visit(payload)
    return result


def extract_paging_state(payload: Any) -> tuple[int | None, bool | None]:
    """응답의 전체 건수와 마지막 페이지 여부를 보수적으로 찾는다.

    목록 API는 요청 size보다 적은 20건을 정상 페이지로 돌려줄 수 있으므로
    페이지 건수만으로 마지막 페이지를 판단하면 안 된다.
    """
    totals: list[int] = []
    last_values: list[bool] = []

    def visit(value: Any) -> None:
        if isinstance(value, dict):
            for key, child in value.items():
                normalized = str(key).replace("_", "").lower()
                if normalized in {"totalcount", "totalelements", "articlecount", "totalarticles"}:
                    try:
                        number = int(child)
                        if number >= 0:
                            totals.append(number)
                    except (TypeError, ValueError):
                        pass
                elif normalized in {"hasmore", "hasnext", "ismore", "ismoredata"}:
                    if isinstance(child, bool):
                        last_values.append(not child)
                elif normalized in {"islast", "last", "lastpage", "isend"}:
                    if isinstance(child, bool):
                        last_values.append(child)
                visit(child)
        elif isinstance(value, list):
            for child in value:
                visit(child)

    visit(payload)
    total = max(totals) if totals else None
    # 하나라도 '더 있음' 응답이면 아직 완료가 아니다.
    is_last = (all(last_values) if last_values else None)
    return total, is_last


def collect_current_articles(naver_realtor_id: str, max_pages: int = 30) -> dict[str, Any]:
    realtor_id = str(naver_realtor_id or "").strip()
    if not realtor_id:
        return {"ok": False, "errors": ["naver_realtor_id_missing"], "articles": []}

    article_nos: list[str] = []
    article_map: dict[str, dict[str, str]] = {}
    response_pages = 0
    completed = False
    declared_total = 0
    errors: list[str] = []

    with sync_playwright() as playwright:
        browser = playwright.chromium.launch(headless=True, args=["--disable-blink-features=AutomationControlled", "--no-sandbox"])
        context = browser.new_context(
            locale="ko-KR", viewport={"width": 1365, "height": 900},
            user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/136.0.0.0 Safari/537.36",
        )
        page = context.new_page()

        def handle_response(response: Any) -> None:
            nonlocal response_pages, completed, declared_total
            if response.url.split("?", 1)[0] != FRONT_API:
                return
            try:
                request_data = json.loads(response.request.post_data or "{}")
                if str(request_data.get("realtorId") or "") != realtor_id:
                    return
                status = int(response.status or 0)
                text = response.text() or ""
                numbers = extract_article_nos(text)
                if status != 200:
                    errors.append(f"front_api_status={status}")
                    return
                if not isinstance(request_data.get("tradeTypes"), list):
                    return
                response_pages += 1
                article_nos.extend(numbers)
                try:
                    payload = json.loads(text)
                    for article_no, item in extract_article_map(payload).items():
                        article_map.setdefault(article_no, {}).update(item)
                    response_total, response_last = extract_paging_state(payload)
                    if response_total is not None:
                        declared_total = max(declared_total, response_total)
                    unique_count = len(uniq(article_nos))
                    # 개별 거래유형 응답의 마지막 페이지를 전체 중개사 매물의
                    # 마지막 페이지로 해석하지 않는다. 실제 완료 판정은 아래의
                    # 연속 무증가 스크롤 검사에서 수행한다.
                except (TypeError, ValueError):
                    pass
                print(
                    f"[MULTI WEBZINE FAST PAGE] page={response_pages} status={status} "
                    f"count={len(numbers)} collected={len(uniq(article_nos))} "
                    f"declared_total={declared_total or '-'} completed={completed}",
                    flush=True,
                )
            except Exception as exc:
                message = str(exc)
                # 정상적인 마지막 페이지 확인 직후 context를 닫는 동안 늦게 도착한
                # 중복 응답 콜백은 수집 실패로 취급하지 않는다.
                if not (
                    completed
                    and article_nos
                    and "Target page, context or browser has been closed" in message
                ):
                    errors.append(f"response_parse:{message}")

        page.on("response", handle_response)
        stable_rounds = 0
        try:
            page.goto(f"https://m.land.naver.com/agency/info/{realtor_id}", wait_until="domcontentloaded", timeout=60000)
            page.wait_for_timeout(2500)
            for _ in range(max(3, min(100, int(max_pages)))):
                before_count = len(uniq(article_nos))
                page.mouse.wheel(0, 6000)
                page.evaluate("""() => { window.scrollTo(0, document.body.scrollHeight); for (const el of document.querySelectorAll('*')) { if (el.scrollHeight > el.clientHeight) el.scrollTop = el.scrollHeight; } }""")
                page.wait_for_timeout(900)
                after_count = len(uniq(article_nos))
                stable_rounds = stable_rounds + 1 if after_count == before_count else 0
                if stable_rounds >= 3 and response_pages > 0:
                    break
        except Exception as exc:
            errors.append(f"browser:{exc}")
        finally:
            # 마지막 response 콜백이 response.text()를 모두 읽을 시간을 확보한다.
            if not page.is_closed():
                page.wait_for_timeout(500)
            page.remove_listener("response", handle_response)
            context.close()
            browser.close()

    numbers = uniq(article_nos)
    completed = bool(numbers) and stable_rounds >= 3
    if not completed:
        errors.append("last_page_not_confirmed")
    if not numbers:
        errors.append("no_articles")
    articles = [
        {
            "article_no": article_no,
            "trade_type": article_map.get(article_no, {}).get("trade_type", ""),
            "registered_at": article_map.get(article_no, {}).get("registered_at", ""),
            "source_rank": rank,
        }
        for rank, article_no in enumerate(numbers)
    ]
    sale = sum(1 for value in numbers if article_map.get(value, {}).get("trade_type") == "A1")
    lease = sum(1 for value in numbers if article_map.get(value, {}).get("trade_type") == "B1")
    rent = sum(1 for value in numbers if article_map.get(value, {}).get("trade_type") == "B2")
    unknown = len(numbers) - sale - lease - rent
    print(f"[MULTI WEBZINE FAST DONE] count={len(numbers)} declared_total={declared_total or '-'} pages={response_pages} sale={sale} lease={lease} rent={rent} unknown={unknown} completed={completed}", flush=True)
    return {
        "ok": completed and bool(numbers) and not errors,
        "naver_realtor_id": realtor_id, "articles": articles, "article_nos": numbers,
        "count": len(numbers), "sale": sale, "lease": lease, "rent": rent,
        "unknown": unknown, "pages": response_pages, "errors": errors,
        "declared_total": declared_total, "complete": completed,
    }


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--naver-realtor-id", required=True)
    parser.add_argument("--max-pages", type=int, default=30)
    args = parser.parse_args()
    result = collect_current_articles(args.naver_realtor_id, args.max_pages)
    print(json.dumps(result, ensure_ascii=False, indent=2))
    if not result["ok"]:
        raise SystemExit(1)


if __name__ == "__main__":
    main()
