import re
from collections import Counter
from datetime import datetime, timedelta

def clean_text(text: str) -> str:
    if not text:
        return ""
    text = re.sub(r"<[^>]+>", " ", text)
    text = re.sub(r"&[a-zA-Z0-9#]+;", " ", text)
    text = re.sub(r"\s+", " ", text)
    return text.strip()

def tokenize_korean_english(text: str):
    if not text:
        return []

    tokens = re.findall(r"[가-힣A-Za-z0-9]{2,}", text)
    stopwords = {
        "그리고", "하지만", "이번", "정말", "너무", "오늘", "내일", "포스팅",
        "블로그", "입니다", "있는", "하는", "위한", "대한", "에서", "으로",
        "with", "this", "that", "have", "your", "about", "from"
    }
    return [t for t in tokens if t.lower() not in stopwords]

def top_keywords_from_posts(posts, top_n=10):
    words = []
    for post in posts:
        text = f"{post.get('title', '')} {post.get('summary', '')}"
        words.extend(tokenize_korean_english(clean_text(text)))

    counter = Counter(words)
    return [{"keyword": k, "count": v} for k, v in counter.most_common(top_n)]

def parse_date_safe(date_str: str):
    if not date_str:
        return None

    s = str(date_str).strip()
    s = re.sub(r"\s+", " ", s).strip()
    now = datetime.now()

    iso_formats = [
        "%Y-%m-%dT%H:%M:%S%z",
        "%Y-%m-%dT%H:%M:%S",
        "%Y-%m-%d %H:%M:%S",
        "%Y-%m-%d %H:%M",
        "%Y-%m-%d",
    ]
    for fmt in iso_formats:
        try:
            return datetime.strptime(s, fmt)
        except Exception:
            pass

    normal_formats = [
        "%Y.%m.%d.",
        "%Y.%m.%d",
        "%Y.%m.%d. %H:%M",
        "%Y.%m.%d %H:%M",
        "%Y. %m. %d.",
        "%Y. %m. %d",
        "%Y. %m. %d. %H:%M",
        "%Y. %m. %d %H:%M",
        "%Y/%m/%d",
        "%Y/%m/%d %H:%M",
        "%Y년 %m월 %d일",
        "%Y년 %m월 %d일 %H:%M",
        "%Y년 %m월 %d일 %H:%M:%S",
    ]
    for fmt in normal_formats:
        try:
            return datetime.strptime(s, fmt)
        except Exception:
            pass

    m = re.match(r"^\s*(\d{4})\s*년\s*(\d{1,2})\s*월\s*(\d{1,2})\s*일(?:\s*(\d{1,2}):(\d{1,2}))?\s*$", s)
    if m:
        year = int(m.group(1))
        month = int(m.group(2))
        day = int(m.group(3))
        hour = int(m.group(4)) if m.group(4) else 0
        minute = int(m.group(5)) if m.group(5) else 0
        try:
            return datetime(year, month, day, hour, minute)
        except Exception:
            pass

    m = re.match(r"^\s*(\d{4})\s*[./-]\s*(\d{1,2})\s*[./-]\s*(\d{1,2})(?:\.)?(?:\s+(\d{1,2}):(\d{1,2}))?\s*$", s)
    if m:
        year = int(m.group(1))
        month = int(m.group(2))
        day = int(m.group(3))
        hour = int(m.group(4)) if m.group(4) else 0
        minute = int(m.group(5)) if m.group(5) else 0
        try:
            return datetime(year, month, day, hour, minute)
        except Exception:
            pass

    m = re.match(r"^\s*(\d+)\s*분\s*전\s*$", s)
    if m:
        return now - timedelta(minutes=int(m.group(1)))

    m = re.match(r"^\s*(\d+)\s*시간\s*전\s*$", s)
    if m:
        return now - timedelta(hours=int(m.group(1)))

    m = re.match(r"^\s*(\d+)\s*일\s*전\s*$", s)
    if m:
        return now - timedelta(days=int(m.group(1)))

    if s == "어제":
        return now - timedelta(days=1)

    if s == "오늘":
        return now

    return None

def calc_avg_post_interval(posts):
    dates = [p["published_at_obj"] for p in posts if p.get("published_at_obj")]
    dates = sorted(dates, reverse=True)

    if len(dates) < 2:
        return 0

    gaps = []
    for i in range(len(dates) - 1):
        gaps.append((dates[i] - dates[i + 1]).days)

    if not gaps:
        return 0

    return round(sum(gaps) / len(gaps), 2)