# -*- coding: utf-8 -*-
"""
STEP105-16 V2 Post Publish URL Resolver

역할:
- V2 최종 발행 후 게시글 URL을 찾는다.
- final_publish 결과의 post_url이 비어 있어도 최신 글 목록/블로그 화면에서
  방금 발행한 제목과 일치하는 URL을 추출한다.
- V1 publish_worker.py는 수정하지 않는다.
"""

import re
import time
from urllib.parse import urlparse, parse_qs


class V2PostPublishUrlResolver:
    def __init__(self, verbose=True):
        self.verbose = verbose

    def resolve(
        self,
        page,
        write_frame=None,
        blog_id=None,
        publish_title=None,
        article_no=None,
        logger=None,
        session=None,
        timeout_seconds=30,
    ):
        def _run():
            return self._resolve_impl(
                page=page,
                write_frame=write_frame,
                blog_id=blog_id,
                publish_title=publish_title,
                article_no=article_no,
                timeout_seconds=timeout_seconds,
            )

        if session and logger:
            with session.step(
                "post_publish.url.resolve",
                logger=logger,
                meta={
                    "blog_id": blog_id,
                    "publish_title": publish_title,
                    "article_no": article_no,
                    "timeout_seconds": timeout_seconds,
                },
            ):
                return _run()

        return _run()

    def _resolve_impl(self, page, write_frame=None, blog_id=None, publish_title=None, article_no=None, timeout_seconds=30):
        blog_id = self._guess_blog_id(page, blog_id=blog_id)
        publish_title = str(publish_title or "").strip()

        result = {
            "status": "not_found",
            "post_url": "",
            "blog_id": blog_id,
            "publish_title": publish_title,
            "article_no": str(article_no or ""),
            "current_url": page.url,
            "candidates": [],
            "steps": [],
        }

        started = time.time()

        candidates = self._collect_candidates_from_targets(page=page, write_frame=write_frame, publish_title=publish_title)
        result["candidates"].extend(candidates)
        result["steps"].append({"step": "current_targets", "count": len(candidates)})

        best = self._pick_best_candidate(candidates, publish_title=publish_title, blog_id=blog_id)
        if best:
            result.update({"status": "success", "post_url": best.get("url") or "", "best": best})
            print("[V2 POST URL RESOLVED current]", result)
            return result

        if blog_id:
            post_list_url = f"https://blog.naver.com/PostList.naver?blogId={blog_id}"
            try:
                print("[V2 POST URL OPEN POSTLIST]", post_list_url)
                page.goto(post_list_url, wait_until="domcontentloaded", timeout=60000)
                page.wait_for_timeout(3000)

                candidates = self._collect_candidates_from_targets(page=page, write_frame=None, publish_title=publish_title)
                result["candidates"].extend(candidates)
                result["steps"].append({"step": "post_list", "url": post_list_url, "count": len(candidates)})

                best = self._pick_best_candidate(candidates, publish_title=publish_title, blog_id=blog_id)
                if best:
                    result.update({
                        "status": "success",
                        "post_url": best.get("url") or "",
                        "best": best,
                        "current_url": page.url,
                    })
                    print("[V2 POST URL RESOLVED post_list]", result)
                    return result
            except Exception as e:
                result["steps"].append({"step": "post_list_error", "error": str(e)})

        if blog_id:
            blog_home = f"https://blog.naver.com/{blog_id}"
            try:
                print("[V2 POST URL OPEN BLOG HOME]", blog_home)
                page.goto(blog_home, wait_until="domcontentloaded", timeout=60000)
                page.wait_for_timeout(3000)

                candidates = self._collect_candidates_from_targets(page=page, write_frame=None, publish_title=publish_title)
                result["candidates"].extend(candidates)
                result["steps"].append({"step": "blog_home", "url": blog_home, "count": len(candidates)})

                best = self._pick_best_candidate(candidates, publish_title=publish_title, blog_id=blog_id)
                if best:
                    result.update({
                        "status": "success",
                        "post_url": best.get("url") or "",
                        "best": best,
                        "current_url": page.url,
                    })
                    print("[V2 POST URL RESOLVED blog_home]", result)
                    return result
            except Exception as e:
                result["steps"].append({"step": "blog_home_error", "error": str(e)})

        result["elapsed_seconds"] = round(time.time() - started, 3)
        result["current_url"] = page.url

        fallback = self._first_post_url_candidate(result["candidates"], blog_id=blog_id)
        if fallback:
            result.update({"status": "fallback", "post_url": fallback.get("url") or "", "best": fallback})

        print("[V2 POST URL RESOLVE RESULT]", result)
        return result

    def _guess_blog_id(self, page, blog_id=None):
        if blog_id:
            return str(blog_id).strip()

        url = str(page.url or "")
        try:
            parsed = urlparse(url)
            qs = parse_qs(parsed.query)

            if "blogId" in qs and qs["blogId"]:
                return qs["blogId"][0]

            if parsed.netloc == "blog.naver.com":
                parts = [x for x in parsed.path.split("/") if x]
                if parts and parts[0] not in ["PostList.naver", "PostView.naver", "GoBlogWrite.naver"]:
                    return parts[0]
        except Exception:
            pass

        return ""

    def _collect_candidates_from_targets(self, page, write_frame=None, publish_title=None):
        targets = [("page", page)]
        if write_frame is not None:
            targets.append(("write_frame", write_frame))

        candidates = []

        for target_name, target in targets:
            try:
                items = target.locator("a[href]").evaluate_all(
                    """els => els.slice(0, 300).map(a => ({
                        href: a.href || '',
                        text: (a.innerText || a.textContent || '').trim(),
                        className: a.getAttribute('class') || '',
                        title: a.getAttribute('title') || '',
                        ariaLabel: a.getAttribute('aria-label') || ''
                    }))"""
                )
            except Exception as e:
                if self.verbose:
                    print("[V2 POST URL COLLECT ERROR]", target_name, str(e))
                items = []

            for item in items:
                href = str(item.get("href") or "").strip()
                if not href or "blog.naver.com" not in href:
                    continue

                text = str(item.get("text") or "").strip()
                title = str(item.get("title") or "").strip()
                aria = str(item.get("ariaLabel") or "").strip()
                cls = str(item.get("className") or "").strip()
                blob = " ".join([text, title, aria])

                candidates.append({
                    "target": target_name,
                    "url": href,
                    "text": text[:300],
                    "title": title[:300],
                    "aria_label": aria[:300],
                    "class": cls[:300],
                    "is_post_like": self._is_post_url(href),
                    "title_score": self._title_score(blob, publish_title),
                })

        dedup = []
        seen = set()
        for c in candidates:
            key = c.get("url")
            if key in seen:
                continue
            seen.add(key)
            dedup.append(c)

        if self.verbose:
            print("[V2 POST URL CANDIDATES]", len(dedup))
            for c in dedup[:20]:
                print(" -", c)

        return dedup

    def _pick_best_candidate(self, candidates, publish_title=None, blog_id=None):
        if not candidates:
            return None

        scored = []
        for c in candidates:
            url = c.get("url") or ""
            score = 0

            if self._is_post_url(url):
                score += 100

            if blog_id and f"blog.naver.com/{blog_id}" in url:
                score += 30

            score += int(c.get("title_score") or 0)

            bad_tokens = [
                "PostList.naver",
                "admin.blog.naver.com",
                "section.blog.naver.com",
                "MyBlog.naver",
                "prologue",
                "guestbook",
                "mapview",
                "library",
                "memo",
                "BlogPrivateTagCloud",
                "Notice.naver",
            ]

            if any(tok in url for tok in bad_tokens):
                score -= 80

            item = dict(c)
            item["score"] = score
            scored.append(item)

        scored = sorted(scored, key=lambda x: x.get("score", 0), reverse=True)
        best = scored[0] if scored else None

        if best and best.get("score", 0) >= 80:
            return best

        if best and publish_title and best.get("title_score", 0) >= 40:
            return best

        return None

    def _first_post_url_candidate(self, candidates, blog_id=None):
        for c in candidates:
            url = c.get("url") or ""
            if self._is_post_url(url):
                if blog_id and f"blog.naver.com/{blog_id}" not in url:
                    continue
                return c
        return None

    def _is_post_url(self, url):
        url = str(url or "")

        if "blog.naver.com" not in url:
            return False

        if "PostView.naver" in url:
            return True

        try:
            parsed = urlparse(url)
            parts = [x for x in parsed.path.split("/") if x]
            if len(parts) >= 2 and parts[-1].isdigit():
                return True
        except Exception:
            pass

        return False

    def _title_score(self, blob, publish_title):
        blob = self._normalize_text(blob)
        title = self._normalize_text(publish_title)

        if not blob or not title:
            return 0

        if title in blob:
            return 100

        title_tokens = [x for x in re.split(r"\s+", title) if len(x) >= 2]
        if not title_tokens:
            return 0

        hit = 0
        for token in title_tokens:
            if token in blob:
                hit += 1

        return int((hit / max(len(title_tokens), 1)) * 80)

    def _normalize_text(self, text):
        return re.sub(r"\s+", " ", str(text or "").strip())


def resolve_post_publish_url(
    page,
    write_frame=None,
    blog_id=None,
    publish_title=None,
    article_no=None,
    logger=None,
    session=None,
    timeout_seconds=30,
    verbose=True,
):
    resolver = V2PostPublishUrlResolver(verbose=verbose)
    return resolver.resolve(
        page=page,
        write_frame=write_frame,
        blog_id=blog_id,
        publish_title=publish_title,
        article_no=article_no,
        logger=logger,
        session=session,
        timeout_seconds=timeout_seconds,
    )
