import re


REGION_TOKENS = [
    "서울", "경기", "인천", "수원", "용인", "성남", "화성", "동탄", "오산",
    "평택", "안양", "안산", "과천", "광명", "시흥", "하남", "남양주", "김포",
    "파주", "고양", "의정부", "대전", "대구", "부산", "광주", "울산", "세종", "제주"
]

PROPERTY_TOKENS = [
    "아파트", "오피스텔", "빌라", "주택", "상가", "토지", "원룸", "투룸", "다가구"
]

TRANSACTION_TOKENS = [
    "매매", "전세", "월세", "분양", "임대", "급매", "투자", "실거주", "매물"
]

INFO_TOKENS = [
    "시세", "입지", "학군", "교통", "생활권", "호재", "실거래가", "비교", "분석", "체크", "정리"
]

CTA_TOKENS = [
    "문의", "상담", "안내", "추천", "체크포인트", "확인사항", "주의사항", "총정리"
]

BRAND_TOKENS = [
    "자이", "래미안", "푸르지오", "더샵", "힐스테이트", "롯데캐슬", "아이파크",
    "e편한", "리슈빌", "센트럴", "파크", "캐슬", "하이츠", "SK뷰", "두산위브", "포레나", "베르디움",
	"디에트르", "금호어울림", "하늘채", "데시앙", "엘리프", "제일풍경채", "서희", "센트레빌", "한신더휴",
	"스위첸", "듀크", "우미", "비발디", "유보라", "뜰", "플래티넘", "빌리브", "애시앙", "펜테리움", "헤링턴", "더리브", "파크드림"
]


def clean_text(text):
    return re.sub(r"\s+", " ", str(text or "")).strip()


def find_token(text, tokens):
    for token in tokens:
        if token in text:
            return token
    return ""


def has_any(text, tokens):
    return any(token in text for token in tokens)


def analyze_title_parts(title):
    title = clean_text(title)

    return {
        "region": find_token(title, REGION_TOKENS),
        "property": find_token(title, PROPERTY_TOKENS),
        "transaction": find_token(title, TRANSACTION_TOKENS),
        "info": find_token(title, INFO_TOKENS),
        "brand": find_token(title, BRAND_TOKENS),
        "has_cta": has_any(title, CTA_TOKENS),
        "length": len(title)
    }


def estimate_title_score(title):
    parts = analyze_title_parts(title)
    score = 35

    if parts["region"]:
        score += 15
    if parts["property"]:
        score += 15
    if parts["transaction"]:
        score += 15
    if parts["info"]:
        score += 10
    if parts["brand"]:
        score += 10
    if parts["has_cta"]:
        score += 8

    if 15 <= parts["length"] <= 38:
        score += 10
    elif parts["length"] < 8:
        score -= 15
    elif parts["length"] > 45:
        score -= 8

    return max(0, min(100, score))


def build_missing_points(title):
    parts = analyze_title_parts(title)
    missing = []

    if not parts["region"]:
        missing.append("지역명")
    if not parts["property"]:
        missing.append("매물유형")
    if not parts["transaction"]:
        missing.append("거래유형")
    if not parts["info"]:
        missing.append("정보형 키워드")
    if not parts["brand"]:
        missing.append("단지명/브랜드명")
    if not parts["has_cta"]:
        missing.append("클릭 유도 표현")

    return missing


def infer_context_from_posts(posts):
    joined = " ".join(clean_text(p.get("title", "")) for p in posts or [])

    region = find_token(joined, REGION_TOKENS) or "지역"
    prop = find_token(joined, PROPERTY_TOKENS) or "아파트"
    txn = find_token(joined, TRANSACTION_TOKENS) or "매매"
    brand = find_token(joined, BRAND_TOKENS)
    info = find_token(joined, INFO_TOKENS) or "시세"

    return {
        "region": region,
        "property": prop,
        "transaction": txn,
        "brand": brand,
        "info": info
    }


def build_click_phrases(region, prop, txn, info):
    return [
        f"{info} 총정리",
        "꼭 볼 체크포인트",
        "상담 전 확인사항",
        "추천 이유",
        "실거래가와 입지 분석",
        "주의사항 정리",
        "비교 분석"
    ]


def build_inserted_keywords(title, posts):
    title = clean_text(title)
    parts = analyze_title_parts(title)
    context = infer_context_from_posts(posts or [])

    region = parts["region"] or context["region"]
    prop = parts["property"] or context["property"]
    txn = parts["transaction"] or context["transaction"]
    brand = parts["brand"] or context["brand"]
    info = parts["info"] or context["info"]

    inserted = []

    if not parts["region"]:
        inserted.append(region)
    if not parts["property"]:
        inserted.append(prop)
    if not parts["transaction"]:
        inserted.append(txn)
    if not parts["info"]:
        inserted.append(info)
    if brand and not parts["brand"]:
        inserted.append(brand)

    return {
        "region": region,
        "property": prop,
        "transaction": txn,
        "brand": brand,
        "info": info,
        "inserted_keywords": list(dict.fromkeys([x for x in inserted if x]))
    }


def build_reason(original_title, new_title, inserted_keywords):
    reasons = []

    if inserted_keywords:
        reasons.append("부족했던 키워드를 제목에 보강했습니다.")

    parts = analyze_title_parts(new_title)

    if parts["region"]:
        reasons.append("지역 검색 유입을 노릴 수 있습니다.")
    if parts["property"]:
        reasons.append("매물 유형이 명확해졌습니다.")
    if parts["transaction"]:
        reasons.append("매매·전세·월세 등 실수요 검색 의도가 분명해졌습니다.")
    if parts["info"]:
        reasons.append("시세·입지·체크포인트 같은 정보형 클릭 요소가 추가되었습니다.")
    if parts["has_cta"]:
        reasons.append("클릭을 유도하는 표현이 포함되었습니다.")

    return reasons[:4]


def generate_keyword_inserted_titles(title, posts=None):
    title = clean_text(title)
    ctx = build_inserted_keywords(title, posts or [])

    region = ctx["region"]
    prop = ctx["property"]
    txn = ctx["transaction"]
    brand = ctx["brand"]
    info = ctx["info"]
    inserted_keywords = ctx["inserted_keywords"]

    click_phrases = build_click_phrases(region, prop, txn, info)

    candidates = []

    if brand:
        candidates.extend([
            f"{region} {brand} {prop} {txn} {info} 총정리",
            f"{region} {brand} {prop} {txn} 꼭 볼 체크포인트",
            f"{region} {brand} {prop} {txn} 상담 전 확인사항",
            f"{region} {brand} {prop} {txn} 실거래가와 입지 분석",
            f"{region} {brand} {prop} {txn} 추천 이유",
        ])
    else:
        candidates.extend([
            f"{region} {prop} {txn} {info} 총정리",
            f"{region} {prop} {txn} 꼭 볼 체크포인트",
            f"{region} {prop} {txn} 상담 전 확인사항",
            f"{region} {prop} {txn} 실거래가와 입지 분석",
            f"{region} {prop} {txn} 추천 이유",
        ])

    # 기존 제목을 살리는 개선형
    if title:
        candidates.extend([
            f"{title} {info} 총정리",
            f"{title} 체크포인트",
            f"{title} 상담 전 확인사항",
        ])

    original_score = estimate_title_score(title)

    results = []
    seen = set()

    for candidate in candidates:
        candidate = clean_text(candidate)

        if not candidate or candidate in seen:
            continue

        seen.add(candidate)

        expected_score = estimate_title_score(candidate)

        results.append({
            "title": candidate,
            "expected_score": expected_score,
            "improvement": max(0, expected_score - original_score),
            "inserted_keywords": inserted_keywords,
            "click_phrase": find_token(candidate, click_phrases) or "",
            "reason": build_reason(title, candidate, inserted_keywords)
        })

    results.sort(key=lambda x: (x["expected_score"], x["improvement"]), reverse=True)

    return results[:5]


def generate_ai_titles_for_post(title, posts=None):
    title = clean_text(title)
    original_score = estimate_title_score(title)
    missing_points = build_missing_points(title)
    suggestions = generate_keyword_inserted_titles(title, posts or [])

    return {
        "original_title": title,
        "original_score": original_score,
        "missing_points": missing_points,
        "suggestions": suggestions
    }


def generate_ai_titles_for_posts(posts, max_items=10):
    output = []

    for post in (posts or [])[:max_items]:
        title = clean_text(post.get("title", ""))

        if not title:
            continue

        output.append({
            "post_url": post.get("post_url", ""),
            "published_at": post.get("published_at", ""),
            "category": post.get("category", ""),
            **generate_ai_titles_for_post(title, posts)
        })

    return output