#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""웹진 전용 네이버 현재 매물번호 수집기.

기존 야간 후보/검색실행/Queue 테이블을 변경하지 않는다. 기존 서비스의
페이지 API 수집 결과를 검증해, 거래유형별 끝 페이지가 확인된 경우에만
현재 매물 Snapshot을 반환한다.
"""

from __future__ import annotations

import argparse
import json
import sys
from pathlib import Path
from typing import Any


FILE_PATH = Path(__file__).resolve()
JOBS_DIR = FILE_PATH.parent
BASE_DIR = JOBS_DIR.parent if JOBS_DIR.name.lower() == "jobs" else JOBS_DIR
for candidate in (JOBS_DIR, BASE_DIR):
    if str(candidate) not in sys.path:
        sys.path.insert(0, str(candidate))

from services.naver_realtor_articles_api import find_realtor_articles_by_api  # type: ignore  # noqa: E402
from services.naver_office_article_finder_playwright import (  # type: ignore  # noqa: E402
    find_articles_by_naver_realtor_id,
)


TRADE_TYPES = ("A1", "B1", "B2")
TRADE_LABELS = {"A1": "sale", "B1": "lease", "B2": "rent"}


def _trade_pages(debug_pages: list[dict[str, Any]], trade_type: str) -> list[dict[str, Any]]:
    pages = [row for row in debug_pages if str(row.get("trade_type") or "") == trade_type]
    pages.sort(key=lambda row: int(row.get("page") or 0))
    return pages


def _trade_complete(pages: list[dict[str, Any]]) -> tuple[bool, str]:
    if not pages or int(pages[0].get("page") or 0) != 1:
        return False, "first_page_missing"
    if any(int(row.get("status") or 0) != 200 for row in pages):
        return False, "non_200_page"
    # 기존 API 수집기는 빈 페이지가 2회 연속이면 해당 거래유형을 종료한다.
    if len(pages) < 2 or any(int(row.get("count") or 0) != 0 for row in pages[-2:]):
        return False, "last_empty_pages_not_confirmed"
    return True, "complete"


def collect_current_articles(naver_realtor_id: str, max_pages: int = 30) -> dict[str, Any]:
    result = find_realtor_articles_by_api(
        str(naver_realtor_id or "").strip(), max_pages=max(3, min(100, int(max_pages)))
    )
    debug_pages = list(result.get("debug_pages") or [])
    article_trade: dict[str, str] = {}
    trade_counts: dict[str, int] = {}
    completion: dict[str, str] = {}
    errors: list[str] = []

    for trade_type in TRADE_TYPES:
        pages = _trade_pages(debug_pages, trade_type)
        complete, reason = _trade_complete(pages)
        completion[trade_type] = reason
        if not complete:
            errors.append(f"{trade_type}:{reason}")
        numbers: set[str] = set()
        for page in pages:
            for value in page.get("article_nos") or []:
                article_no = str(value or "").strip()
                if article_no:
                    numbers.add(article_no)
                    article_trade[article_no] = trade_type
        trade_counts[TRADE_LABELS[trade_type]] = len(numbers)

    # 거래유형별 결과를 기준으로 사용한다. 전체형(빈 tradeType)은 검증·보조 로그용이다.
    articles = [
        {"article_no": article_no, "trade_type": trade_type}
        for article_no, trade_type in sorted(article_trade.items())
    ]
    ok = bool(result.get("ok")) and not errors and bool(articles)
    diagnostic: dict[str, Any] = {}
    if not ok:
        # 현재 네이버가 직접 /api/articles 호출을 401로 거부하는 경우,
        # 실제 화면이 사용하는 요청 URL/POST DATA를 기존 읽기 전용 수집기로
        # 포착한다. 이 호출은 DB·후보·Queue를 변경하지 않는다.
        print("[MULTI WEBZINE API FALLBACK] capturing_browser_requests", flush=True)
        browser_result = find_articles_by_naver_realtor_id(
            str(naver_realtor_id or "").strip(), headless=True
        )
        source_counts = [
            len(source.get("article_nos") or [])
            for source in (browser_result.get("debug_sources") or [])
        ]
        captured_urls: list[str] = []
        for source in browser_result.get("debug_sources") or []:
            for url in source.get("network_urls") or []:
                text_url = str(url or "")
                if (
                    text_url
                    and ("article" in text_url or "realtor" in text_url or "agency" in text_url)
                    and text_url not in captured_urls
                ):
                    captured_urls.append(text_url)
        diagnostic = {
            "browser_candidate_count": len(browser_result.get("candidates") or []),
            "source_counts": source_counts,
            "captured_urls": captured_urls,
        }
        print(
            "[MULTI WEBZINE API DIAGNOSTIC] "
            f"browser_candidates={diagnostic['browser_candidate_count']} "
            f"source_counts={source_counts}",
            flush=True,
        )
        for url in captured_urls:
            print(f"[MULTI WEBZINE CAPTURED URL] {url}", flush=True)
    return {
        "ok": ok,
        "naver_realtor_id": str(naver_realtor_id or "").strip(),
        "articles": articles,
        "article_nos": [row["article_no"] for row in articles],
        "count": len(articles),
        "sale": trade_counts.get("sale", 0),
        "lease": trade_counts.get("lease", 0),
        "rent": trade_counts.get("rent", 0),
        "completion": completion,
        "errors": errors,
        "diagnostic": diagnostic,
    }


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--naver-realtor-id", required=True)
    parser.add_argument("--max-pages", type=int, default=30)
    args = parser.parse_args()
    result = collect_current_articles(args.naver_realtor_id, args.max_pages)
    print(json.dumps(result, ensure_ascii=False, indent=2))
    if not result["ok"]:
        raise SystemExit(1)


if __name__ == "__main__":
    main()
