import re


REGION_TOKENS = [
    "서울", "경기", "인천", "수원", "용인", "성남", "화성", "동탄", "오산",
    "평택", "안양", "안산", "과천", "광명", "시흥", "하남", "남양주", "김포",
    "파주", "고양", "의정부", "대전", "대구", "부산", "광주", "울산", "세종", "제주"
]

PROPERTY_TOKENS = [
    "아파트", "오피스텔", "빌라", "주택", "상가", "토지", "원룸", "투룸", "다가구"
]

TRANSACTION_TOKENS = [
    "매매", "전세", "월세", "분양", "임대", "급매", "투자", "실거주", "매물"
]

INFO_TOKENS = [
    "시세", "실거래가", "입지", "학군", "교통", "호재", "생활권",
    "분석", "비교", "체크", "정리", "가격", "전망"
]

CONSULTING_TOKENS = [
    "문의", "상담", "연락", "안내", "확인사항", "체크포인트", "방문", "예약"
]

BRAND_TOKENS = [
    "자이", "래미안", "푸르지오", "더샵", "힐스테이트", "롯데캐슬", "아이파크",
    "e편한", "리슈빌", "센트럴", "파크", "캐슬", "하이츠", "SK뷰", "두산위브", "포레나", "베르디움",
	"디에트르", "금호어울림", "하늘채", "데시앙", "엘리프", "제일풍경채", "서희", "센트레빌", "한신더휴",
	"스위첸", "듀크", "우미", "비발디", "유보라", "뜰", "플래티넘", "빌리브", "애시앙", "펜테리움", "헤링턴", "더리브", "파크드림"
]


def clean_text(text):
    return re.sub(r"\s+", " ", str(text or "")).strip()


def has_any(text, tokens):
    text = clean_text(text)
    return any(token in text for token in tokens)


def count_posts_with(posts, tokens):
    count = 0
    for post in posts or []:
        title = clean_text(post.get("title", ""))
        if has_any(title, tokens):
            count += 1
    return count


def classify_intent(title):
    title = clean_text(title)

    listing = has_any(title, PROPERTY_TOKENS) and has_any(title, TRANSACTION_TOKENS)
    info = has_any(title, INFO_TOKENS)
    consulting = has_any(title, CONSULTING_TOKENS)
    local = has_any(title, REGION_TOKENS)
    brand = has_any(title, BRAND_TOKENS)

    if listing:
        return "listing"
    if info:
        return "info"
    if consulting:
        return "consulting"
    if local or brand:
        return "local_branding"

    return "etc"


def grade(score):
    score = float(score or 0)

    if score >= 85:
        return "A+"
    if score >= 75:
        return "A"
    if score >= 60:
        return "B"
    if score >= 45:
        return "C"
    return "D"


def build_naver_seo_score(posts, analysis=None):
    posts = posts or []
    analysis = analysis or {}
    total = len(posts)

    if total == 0:
        return {
            "crank_score": 0,
            "crank_grade": "D",
            "dia_score": 0,
            "dia_grade": "D",
            "overall_score": 0,
            "overall_grade": "D",
            "intent_mix": {
                "listing": 0,
                "info": 0,
                "consulting": 0,
                "local_branding": 0,
                "etc": 0
            },
            "summary": "게시글 데이터가 부족해 네이버 최적화 진단을 수행하기 어렵습니다.",
            "comments": ["게시글 수집 결과가 없습니다."],
            "actions": ["블로그 공개 상태와 게시글 수집 상태를 먼저 확인하세요."]
        }

    region_count = count_posts_with(posts, REGION_TOKENS)
    property_count = count_posts_with(posts, PROPERTY_TOKENS)
    transaction_count = count_posts_with(posts, TRANSACTION_TOKENS)
    info_count = count_posts_with(posts, INFO_TOKENS)
    consulting_count = count_posts_with(posts, CONSULTING_TOKENS)
    brand_count = count_posts_with(posts, BRAND_TOKENS)

    recent_30d = int(analysis.get("recent_30d_posts", 0) or 0)
    top_keywords = analysis.get("top_keywords", []) or []

    intent_mix = {
        "listing": 0,
        "info": 0,
        "consulting": 0,
        "local_branding": 0,
        "etc": 0
    }

    dia_post_scores = []

    for post in posts:
        title = clean_text(post.get("title", ""))
        intent = classify_intent(title)
        intent_mix[intent] += 1

        score = 30

        if has_any(title, REGION_TOKENS):
            score += 15
        if has_any(title, PROPERTY_TOKENS):
            score += 15
        if has_any(title, TRANSACTION_TOKENS):
            score += 15
        if has_any(title, INFO_TOKENS):
            score += 15
        if has_any(title, CONSULTING_TOKENS):
            score += 5
        if has_any(title, BRAND_TOKENS):
            score += 5

        title_len = len(title)
        if 15 <= title_len <= 40:
            score += 10
        elif title_len < 8:
            score -= 10
        elif title_len > 50:
            score -= 8

        dia_post_scores.append(max(0, min(100, score)))

    dia_score = round(sum(dia_post_scores) / len(dia_post_scores), 2) if dia_post_scores else 0

    crank_score = 25
    crank_score += min(region_count / total * 20, 20)
    crank_score += min(property_count / total * 18, 18)
    crank_score += min(transaction_count / total * 15, 15)
    crank_score += min(info_count / total * 15, 15)
    crank_score += min(brand_count / total * 10, 10)

    if recent_30d >= 8:
        crank_score += 12
    elif recent_30d >= 4:
        crank_score += 8
    elif recent_30d >= 1:
        crank_score += 4

    if len(top_keywords) >= 8:
        crank_score += 5
    elif len(top_keywords) >= 4:
        crank_score += 3

    crank_score = round(max(0, min(100, crank_score)), 2)

    overall_score = round((crank_score * 0.45) + (dia_score * 0.55), 2)

    comments = []
    actions = []

    if region_count > 0:
        comments.append("지역 키워드가 포함되어 지역 기반 검색 전문성 형성에 도움이 됩니다.")
    else:
        comments.append("지역 키워드가 부족해 지역 전문성 신호가 약할 수 있습니다.")
        actions.append("제목과 본문에 동네명, 역명, 단지명 등 지역 키워드를 꾸준히 포함하세요.")

    if property_count > 0:
        comments.append("매물 유형 키워드가 포함되어 부동산 주제성이 비교적 명확합니다.")
    else:
        actions.append("아파트, 오피스텔, 상가, 빌라 등 매물 유형 키워드를 제목에 반영하세요.")

    if transaction_count > 0:
        comments.append("매매·전세·월세 등 거래 의도가 포함되어 실수요 검색 대응에 유리합니다.")
    else:
        actions.append("매매, 전세, 월세, 급매, 분양 같은 거래유형 키워드를 보강하세요.")

    if info_count > 0:
        comments.append("시세·입지·실거래가 등 정보형 콘텐츠 요소가 포함되어 DIA 문서 품질에 긍정적입니다.")
    else:
        comments.append("정보형 키워드가 부족해 검색 의도 대응력이 약할 수 있습니다.")
        actions.append("시세, 실거래가, 입지, 학군, 교통, 비교 분석형 글을 주기적으로 작성하세요.")

    if intent_mix["listing"] >= total * 0.8:
        comments.append("매물형 콘텐츠 비중이 높아 정보형 유입 확장에는 제한이 있을 수 있습니다.")
        actions.append("매물 소개 글 외에 지역 분석, 시세 분석, 학군/교통 분석 글을 섞어 운영하세요.")

    if intent_mix["info"] == 0:
        actions.append("정보형 콘텐츠가 부족합니다. ‘실거래가 정리’, ‘입지 분석’, ‘시세 비교’ 글을 추가하세요.")

    if consulting_count == 0:
        actions.append("상담 전 확인사항, 문의 안내, 체크포인트 같은 전환 유도 문구를 자연스럽게 포함하세요.")

    if recent_30d <= 1:
        actions.append("최근 발행량이 부족합니다. 최소 주 1~2회 이상 발행 리듬을 유지하세요.")

    if overall_score >= 75:
        summary = "네이버 검색 최적화 관점에서 비교적 양호한 상태입니다. 정보형 콘텐츠와 상담 유도 구조를 더하면 성과가 좋아질 수 있습니다."
    elif overall_score >= 60:
        summary = "기본적인 검색 최적화 요소는 있으나, 지역성·정보성·상담 전환 구조를 더 보강할 필요가 있습니다."
    elif overall_score >= 45:
        summary = "네이버 검색 최적화 적합도가 다소 약합니다. 지역명, 매물유형, 거래유형, 정보형 콘텐츠를 우선 보강해야 합니다."
    else:
        summary = "검색 최적화 기반이 약한 편입니다. 부동산 전문성 신호와 문서 품질 요소를 체계적으로 다시 구성해야 합니다."

    if not actions:
        actions.append("현재 강점을 유지하면서 정보형 콘텐츠와 상담 유도 문구를 꾸준히 확장하세요.")

    return {
        "crank_score": crank_score,
        "crank_grade": grade(crank_score),
        "dia_score": dia_score,
        "dia_grade": grade(dia_score),
        "overall_score": overall_score,
        "overall_grade": grade(overall_score),
        "intent_mix": intent_mix,
        "metrics": {
            "total_posts": total,
            "region_count": region_count,
            "property_count": property_count,
            "transaction_count": transaction_count,
            "info_count": info_count,
            "consulting_count": consulting_count,
            "brand_count": brand_count,
            "recent_30d_posts": recent_30d
        },
        "summary": summary,
        "comments": comments[:6],
        "actions": actions[:6]
    }