import os
import re
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from config import REQUEST_TIMEOUT

HEADERS = {
    "User-Agent": "Mozilla/5.0"
}

DATE_REGEXES = [
    r"\d{4}\.\s*\d{1,2}\.\s*\d{1,2}\.\s*\d{1,2}:\d{2}",
    r"\d{4}\.\s*\d{1,2}\.\s*\d{1,2}\.",
    r"\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}",
    r"\d{4}-\d{2}-\d{2} \d{2}:\d{2}:\d{2}",
    r"\d{4}-\d{2}-\d{2}",
    r"\d{4}년\s*\d{1,2}월\s*\d{1,2}일(?:\s*\d{1,2}:\d{2})?",
    r"\d+\s*분\s*전",
    r"\d+\s*시간\s*전",
    r"\d+\s*일\s*전",
    r"어제",
    r"오늘",
]

POST_URL_PATTERNS = [
    r"https?://blog\.naver\.com/[A-Za-z0-9._-]+/\d+",
    r"https?://blog\.naver\.com/PostView\.naver\?[^\"' <>\n]+",
    r"/PostView\.naver\?[^\"' <>\n]+",
    r"/[A-Za-z0-9._-]+/\d+",
]

def fetch_url(url: str):
    response = requests.get(url, headers=HEADERS, timeout=REQUEST_TIMEOUT)
    response.raise_for_status()
    return response.text

def fetch_blog_main(blog_id: str):
    url = f"https://blog.naver.com/{blog_id}"
    html = fetch_url(url)
    return url, html

def fetch_blog_info(blog_id: str):
    url, html = fetch_blog_main(blog_id)
    soup = BeautifulSoup(html, "html.parser")
    title = soup.title.get_text(strip=True) if soup.title else blog_id

    return {
        "blog_id": blog_id,
        "blog_url": url,
        "blog_name": title
    }

def extract_iframe_src(html: str):
    soup = BeautifulSoup(html, "html.parser")
    iframe = soup.select_one("iframe#mainFrame")
    return iframe.get("src") if iframe and iframe.get("src") else None

def normalize_url(url: str, base_url: str):
    full_url = urljoin(base_url, url)
    return full_url.strip()

def is_post_url(url: str):
    return (
        "/PostView.naver" in url
        or re.search(r"blog\.naver\.com/[A-Za-z0-9._-]+/\d+$", url) is not None
    )

def dedupe_urls(urls):
    dedup = []
    seen = set()
    for url in urls:
        url = url.strip()
        if not url:
            continue
        if url not in seen:
            seen.add(url)
            dedup.append(url)
    return dedup

def extract_post_links_from_html(html: str, base_url: str):
    soup = BeautifulSoup(html, "html.parser")
    results = []

    # 1차: a href 기반
    for a in soup.select("a[href]"):
        href = a.get("href", "").strip()
        if not href:
            continue
        full_url = normalize_url(href, base_url)
        if "blog.naver.com" in full_url and is_post_url(full_url):
            results.append(full_url)

    # 2차: HTML 전체 정규식 스캔
    compact_html = re.sub(r"\s+", " ", html)
    for pattern in POST_URL_PATTERNS:
        matches = re.findall(pattern, compact_html)
        for m in matches:
            full_url = normalize_url(m, base_url)
            if "blog.naver.com" in full_url and is_post_url(full_url):
                results.append(full_url)

    results = dedupe_urls(results)

    print("[extract_post_links_from_html] base_url:", base_url)
    print("[extract_post_links_from_html] found:", len(results))
    for i, url in enumerate(results[:20], start=1):
        print(f"[extract_post_links_from_html] {i}:", url)

    return results

def extract_text_by_selectors(soup, selectors):
    for selector in selectors:
        try:
            nodes = soup.select(selector)
            for node in nodes:
                text = node.get_text(" ", strip=True)
                if text:
                    return text
        except Exception:
            continue
    return ""

def extract_meta_content(soup, selectors):
    for selector in selectors:
        try:
            node = soup.select_one(selector)
            if node:
                content = node.get("content", "").strip()
                if content:
                    return content
        except Exception:
            continue
    return ""

def normalize_found_date(value: str):
    if not value:
        return ""
    return re.sub(r"\s+", " ", value).strip()

def extract_date_from_html_text(html: str):
    compact_html = re.sub(r"\s+", " ", html)
    for pattern in DATE_REGEXES:
        m = re.search(pattern, compact_html)
        if m:
            return normalize_found_date(m.group(0))
    return ""

def save_debug_html(name: str, html: str):
    os.makedirs("debug_pages", exist_ok=True)
    safe_name = re.sub(r"[^A-Za-z0-9._-]", "_", name)
    filepath = os.path.join("debug_pages", f"{safe_name[:120]}.html")
    with open(filepath, "w", encoding="utf-8") as f:
        f.write(html)
    print("[save_debug_html] SAVED:", filepath)
    return filepath

def fetch_post_detail(post_url: str):
    html = fetch_url(post_url)
    save_debug_html(post_url, html)

    soup = BeautifulSoup(html, "html.parser")

    title = extract_text_by_selectors(soup, [
        ".se-title-text span",
        ".se-title-text",
        ".pcol1 .title_1",
        ".htitle .pcol1",
        ".se-module.se-module-text .se-title-text",
        ".tit_area .title",
        ".post_title",
        "h3",
        "h2"
    ])

    if not title:
        title = extract_meta_content(soup, [
            "meta[property='og:title']",
            "meta[name='title']"
        ])

    summary = extract_meta_content(soup, [
        "meta[property='og:description']",
        "meta[name='description']"
    ])

    published_at = extract_text_by_selectors(soup, [
        ".se_publishDate",
        "span.se_publishDate",
        ".post_write_time",
        ".blog2_series_date",
        ".wrap_blog2_entry_info .date",
        ".blog_date",
        ".date",
        ".publish_date",
        "time"
    ])

    if not published_at:
        published_at = extract_meta_content(soup, [
            "meta[property='article:published_time']",
            "meta[name='article:published_time']",
            "meta[property='og:updated_time']",
            "meta[property='article:modified_time']"
        ])

    if not published_at:
        published_at = extract_date_from_html_text(html)

    category = extract_text_by_selectors(soup, [
        ".blog2_series a",
        ".cate_area a",
        ".post_category a",
        ".category a",
        ".blog_category a"
    ])

    published_at = normalize_found_date(published_at)

    print("=" * 80)
    print("[fetch_post_detail] URL:", post_url)
    print("[fetch_post_detail] TITLE:", title)
    print("[fetch_post_detail] PUBLISHED_AT:", published_at)
    print("[fetch_post_detail] CATEGORY:", category)

    return {
        "post_url": post_url,
        "title": title,
        "summary": summary,
        "published_at": published_at,
        "category": category
    }

def fetch_posts(blog_id: str, limit: int = 10):
    base_url, html = fetch_blog_main(blog_id)
    save_debug_html(f"{blog_id}_main", html)

    links = extract_post_links_from_html(html, base_url)

    iframe_src = extract_iframe_src(html)
    if iframe_src:
        try:
            iframe_url = urljoin(base_url, iframe_src)
            iframe_html = fetch_url(iframe_url)
            save_debug_html(f"{blog_id}_iframe", iframe_html)

            iframe_links = extract_post_links_from_html(iframe_html, iframe_url)
            links.extend(iframe_links)
        except Exception as e:
            print("[fetch_posts] iframe load error:", str(e))

    links = dedupe_urls(links)

    print("[fetch_posts] TOTAL LINK CANDIDATES:", len(links))
    for i, url in enumerate(links[:30], start=1):
        print(f"[fetch_posts] LINK {i}:", url)

    posts = []
    for url in links[:limit * 8]:
        try:
            post = fetch_post_detail(url)
            if post.get("title"):
                posts.append(post)
            if len(posts) >= limit:
                break
        except Exception as e:
            print("[fetch_posts] ERROR:", str(e))
            continue

    print("[fetch_posts] FINAL COUNT:", len(posts))
    return posts