# -*- coding: utf-8 -*-
"""
V2-015 ExposureChecker v1

노출 판정 기준:
- 오늘 발행한 네이버 블로그 포스트 제목을 네이버 검색에서 검색한다.
- 검색 영역은 블로그 탭 기준이다.
- 정렬은 최신순 기준이다.
- 첫 페이지에서 해당 blog_url 또는 blog_post_no가 확인되면 "노출"로 판정한다.

주의:
- 이 v1 파일은 판정 기준과 URL 생성/결과 파싱 골격만 담당한다.
- 실제 Playwright 자동 검색 연결은 다음 단계에서 붙인다.
"""

import re
from dataclasses import dataclass
from urllib.parse import quote_plus


@dataclass
class ExposureCheckTarget:
    history_id: int | None = None
    realtor_id: int | None = None
    article_no: str | None = None
    blog_url: str | None = None
    blog_id: str | None = None
    blog_post_no: str | None = None
    publish_title: str | None = None
    search_keyword: str | None = None


@dataclass
class ExposureCheckResult:
    success: bool
    is_exposed: bool
    search_status: str
    message: str = ""
    exposure_rank: int | None = None
    exposure_page: int | None = None
    search_url: str | None = None
    matched_url: str | None = None


class ExposureChecker:
    """
    네이버 검색 > 블로그 > 최신순 > 첫 페이지 노출 여부를 판정하는 클래스.
    """

    def normalize_blog_post_no(self, value):
        value = str(value or "").strip()
        if not value:
            return ""

        match = re.search(r"/(\d{8,})", value)
        if match:
            return match.group(1)

        match = re.search(r"logNo=(\d{8,})", value)
        if match:
            return match.group(1)

        if re.fullmatch(r"\d{8,}", value):
            return value

        return ""

    def normalize_blog_url(self, value):
        value = str(value or "").strip()
        if not value:
            return ""

        value = value.replace("m.blog.naver.com", "blog.naver.com")
        value = value.split("?")[0].rstrip("/")
        return value

    def keyword_from_target(self, target: ExposureCheckTarget) -> str:
        keyword = (target.search_keyword or target.publish_title or "").strip()
        return re.sub(r"\s+", " ", keyword)

    def build_search_url(self, keyword: str, page: int = 1) -> str:
        """
        네이버 검색 블로그 탭 최신순 URL.
        nso=so:dd 는 최신순 정렬에 해당하는 형태로 사용한다.
        """
        start = 1 if page <= 1 else ((page - 1) * 10 + 1)
        encoded = quote_plus(keyword)

        return (
            "https://search.naver.com/search.naver"
            f"?where=blog"
            f"&query={encoded}"
            f"&sm=tab_opt"
            f"&nso=so:dd,p:all"
            f"&start={start}"
        )

    def match_item(self, target: ExposureCheckTarget, candidate_url: str, candidate_text: str = "") -> bool:
        """
        검색 결과 1개가 대상 포스트인지 판정한다.
        blog_post_no 또는 blog_url 기준으로 매칭한다.
        """
        candidate_url_norm = self.normalize_blog_url(candidate_url)
        target_url_norm = self.normalize_blog_url(target.blog_url)

        target_post_no = (
            self.normalize_blog_post_no(target.blog_post_no)
            or self.normalize_blog_post_no(target.blog_url)
        )

        candidate_post_no = self.normalize_blog_post_no(candidate_url_norm)

        if target_post_no and candidate_post_no and target_post_no == candidate_post_no:
            return True

        if target_url_norm and candidate_url_norm:
            if candidate_url_norm == target_url_norm:
                return True

            if target_url_norm in candidate_url_norm or candidate_url_norm in target_url_norm:
                return True

        return False

    def check_from_candidates(self, target: ExposureCheckTarget, candidates: list[dict]) -> ExposureCheckResult:
        """
        Playwright 연결 전 테스트용.
        candidates 예:
        [
            {"url": "...", "title": "..."},
            ...
        ]
        """
        keyword = self.keyword_from_target(target)
        search_url = self.build_search_url(keyword, page=1)

        for idx, item in enumerate(candidates or [], start=1):
            url = item.get("url") or item.get("href") or ""
            title = item.get("title") or item.get("text") or ""

            if self.match_item(target, url, title):
                return ExposureCheckResult(
                    success=True,
                    is_exposed=True,
                    search_status="exposed",
                    message="exposed on first blog search page",
                    exposure_rank=idx,
                    exposure_page=1,
                    search_url=search_url,
                    matched_url=url,
                )

        return ExposureCheckResult(
            success=True,
            is_exposed=False,
            search_status="not_exposed",
            message="not found on first blog search page",
            exposure_rank=None,
            exposure_page=1,
            search_url=search_url,
            matched_url=None,
        )
