import re


REGION_TOKENS = [
    "서울", "경기", "인천", "수원", "용인", "성남", "화성", "동탄", "오산",
    "평택", "안양", "안산", "과천", "광명", "시흥", "하남", "남양주", "김포",
    "파주", "고양", "의정부", "대전", "대구", "부산", "광주", "울산", "세종", "제주"
]

PROPERTY_TOKENS = [
    "아파트", "오피스텔", "빌라", "주택", "상가", "토지", "원룸", "투룸", "다가구"
]

TRANSACTION_TOKENS = [
    "매매", "전세", "월세", "분양", "임대", "급매", "투자", "실거주", "매물"
]

INFO_TOKENS = [
    "시세", "학군", "입지", "교통", "생활권", "호재", "비교", "분석", "체크", "정리", "추천"
]

APT_BRAND_TOKENS = [
    "자이", "래미안", "푸르지오", "더샵", "힐스테이트", "롯데캐슬", "아이파크",
    "e편한", "리슈빌", "센트럴", "파크", "캐슬", "하이츠", "SK뷰", "두산위브", "포레나", "베르디움",
	"디에트르", "금호어울림", "하늘채", "데시앙", "엘리프", "제일풍경채", "서희", "센트레빌", "한신더휴",
	"스위첸", "듀크", "우미", "비발디", "유보라", "뜰", "플래티넘", "빌리브", "애시앙", "펜테리움", "헤링턴", "더리브", "파크드림"
]


def clean_spaces(text: str) -> str:
    return re.sub(r"\s+", " ", text or "").strip()


def extract_first_token(text: str, tokens: list[str]) -> str:
    for token in tokens:
        if token in text:
            return token
    return ""


def analyze_title_pattern(title: str) -> dict:
    t = clean_spaces(title)

    region = extract_first_token(t, REGION_TOKENS)
    prop = extract_first_token(t, PROPERTY_TOKENS)
    txn = extract_first_token(t, TRANSACTION_TOKENS)
    info = extract_first_token(t, INFO_TOKENS)
    brand = extract_first_token(t, APT_BRAND_TOKENS)

    return {
        "title": t,
        "has_region": bool(region),
        "has_property": bool(prop),
        "has_transaction": bool(txn),
        "has_info": bool(info),
        "has_brand": bool(brand),
        "region": region,
        "property": prop,
        "transaction": txn,
        "info": info,
        "brand": brand,
        "length": len(t),
    }


def summarize_patterns(posts: list[dict], max_items: int = 10) -> dict:
    titles = [clean_spaces(p.get("title", "")) for p in (posts or [])[:max_items] if clean_spaces(p.get("title", ""))]
    patterns = [analyze_title_pattern(t) for t in titles]

    if not patterns:
        return {
            "count": 0,
            "avg_length": 0,
            "region_count": 0,
            "property_count": 0,
            "transaction_count": 0,
            "info_count": 0,
            "brand_count": 0,
            "top_regions": [],
            "top_transactions": [],
            "top_properties": [],
            "sample_titles": [],
        }

    def top_values(key: str, limit: int = 3):
        counts = {}
        for p in patterns:
            val = p.get(key, "")
            if val:
                counts[val] = counts.get(val, 0) + 1
        return sorted(counts.items(), key=lambda x: (-x[1], x[0]))[:limit]

    return {
        "count": len(patterns),
        "avg_length": round(sum(p["length"] for p in patterns) / len(patterns), 2),
        "region_count": sum(1 for p in patterns if p["has_region"]),
        "property_count": sum(1 for p in patterns if p["has_property"]),
        "transaction_count": sum(1 for p in patterns if p["has_transaction"]),
        "info_count": sum(1 for p in patterns if p["has_info"]),
        "brand_count": sum(1 for p in patterns if p["has_brand"]),
        "top_regions": top_values("region"),
        "top_transactions": top_values("transaction"),
        "top_properties": top_values("property"),
        "sample_titles": titles[:5],
    }


def pattern_comment(base_summary: dict, comp_summary: dict, comp_blog_id: str) -> list[str]:
    comments = []

    if comp_summary["region_count"] > base_summary["region_count"]:
        comments.append(f"{comp_blog_id}는 제목에 지역명을 더 자주 넣고 있습니다.")

    if comp_summary["transaction_count"] > base_summary["transaction_count"]:
        comments.append(f"{comp_blog_id}는 매매·전세·월세 같은 거래형 키워드를 더 자주 사용합니다.")

    if comp_summary["property_count"] > base_summary["property_count"]:
        comments.append(f"{comp_blog_id}는 아파트·오피스텔 등 매물유형 표현이 더 분명합니다.")

    if comp_summary["info_count"] > base_summary["info_count"]:
        comments.append(f"{comp_blog_id}는 시세·입지·분석형 제목 비중이 더 높습니다.")

    if comp_summary["brand_count"] > base_summary["brand_count"]:
        comments.append(f"{comp_blog_id}는 단지명 또는 브랜드명 중심 제목을 더 자주 사용합니다.")

    if comp_summary["avg_length"] > base_summary["avg_length"] + 4:
        comments.append(f"{comp_blog_id}는 제목 길이가 조금 더 길고 구체적인 편입니다.")

    return comments[:5]


def generate_pattern_based_titles(base_posts: list[dict], comp_posts: list[dict], max_items: int = 3) -> list[str]:
    base_titles = [clean_spaces(p.get("title", "")) for p in (base_posts or []) if clean_spaces(p.get("title", ""))]
    comp_titles = [clean_spaces(p.get("title", "")) for p in (comp_posts or []) if clean_spaces(p.get("title", ""))]

    base_join = " ".join(base_titles)
    comp_join = " ".join(comp_titles)

    region = extract_first_token(base_join, REGION_TOKENS) or extract_first_token(comp_join, REGION_TOKENS) or "지역"
    prop = extract_first_token(base_join, PROPERTY_TOKENS) or extract_first_token(comp_join, PROPERTY_TOKENS) or "아파트"
    txn = extract_first_token(base_join, TRANSACTION_TOKENS) or extract_first_token(comp_join, TRANSACTION_TOKENS) or "매매"
    info = extract_first_token(comp_join, INFO_TOKENS) or "시세"
    brand = extract_first_token(base_join, APT_BRAND_TOKENS) or extract_first_token(comp_join, APT_BRAND_TOKENS)

    suggestions = []

    if brand:
        suggestions.append(f"{region} {brand} {prop} {txn} {info} 총정리")
        suggestions.append(f"{region} {brand} {prop} {txn} 체크포인트")
        suggestions.append(f"{region} {brand} {prop} {txn} 추천 이유")
    else:
        suggestions.append(f"{region} {prop} {txn} {info} 총정리")
        suggestions.append(f"{region} {prop} {txn} 입지와 체크포인트")
        suggestions.append(f"{region} {prop} {txn} 추천 이유와 비교")

    out = []
    seen = set()
    for s in suggestions:
        s = clean_spaces(s)
        if s and s not in seen:
            seen.add(s)
            out.append(s)

    return out[:max_items]


def build_title_pattern_insights(results: list[dict], base_blog_id: str = "") -> list[dict]:
    successful = [x for x in results if x.get("success")]
    if len(successful) < 2:
        return []

    base = None

    if base_blog_id:
        for item in successful:
            if item.get("blog", {}).get("blog_id", "") == base_blog_id:
                base = item
                break

    if base is None:
        base = successful[0]

    base_posts = base.get("posts", []) or []
    base_summary = summarize_patterns(base_posts, 10)
    base_blog_id_final = base.get("blog", {}).get("blog_id", "")

    competitors = [
        x for x in successful
        if x.get("blog", {}).get("blog_id", "") != base_blog_id_final
    ]

    insights = []

    for comp in competitors:
        comp_blog_id = comp.get("blog", {}).get("blog_id", "")
        comp_posts = comp.get("posts", []) or []
        comp_summary = summarize_patterns(comp_posts, 10)

        insights.append({
            "base_blog_id": base_blog_id_final,
            "competitor_blog_id": comp_blog_id,
            "base_pattern_summary": base_summary,
            "competitor_pattern_summary": comp_summary,
            "pattern_comments": pattern_comment(base_summary, comp_summary, comp_blog_id),
            "recommended_titles": generate_pattern_based_titles(base_posts, comp_posts, 3),
        })

    return insights