# -*- coding: utf-8 -*-
"""
STEP111-03 HEE Available Data Analyzer v1

목적:
- 네이버부동산 원천 데이터 기준으로 "작성 가능한 범위"를 판단한다.
- 정보가 적다고 실패 처리하지 않는다.
- 사진이 없거나 1장뿐인 매물도 정상 케이스로 인정한다.
- 단, 사진이 부족한데 사진 중심 레이아웃을 쓰는 것은 막는다.
- 토지/공장/창고/상가/사무실 등 비아파트 매물 특성을 감안한다.

사용:
  cd /d D:\honghee\blog_api

  draft 기준:
  python tools\hee_available_data_analyzer.py --draft-id 209

  article 기준 최신 draft:
  python tools\hee_available_data_analyzer.py --realtor-id 1 --article-no 2633019770

  JSON:
  python tools\hee_available_data_analyzer.py --draft-id 209 --json
"""

import argparse
import json
import re
import sys
from pathlib import Path
from html import unescape

ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT))

from db import get_conn


def clean_text(value):
    value = "" if value is None else str(value)
    value = unescape(value)
    value = re.sub(r"<[^>]+>", " ", value)
    value = re.sub(r"\s+", " ", value)
    return value.strip()


def safe_json_loads(value):
    if not value:
        return {}
    if isinstance(value, dict):
        return value
    try:
        return json.loads(value)
    except Exception:
        return {}


def fetch_draft(conn, draft_id=None, realtor_id=None, article_no=None):
    with conn.cursor() as cur:
        if draft_id:
            cur.execute("SELECT * FROM blog_article_drafts WHERE id=%s LIMIT 1", (int(draft_id),))
            return cur.fetchone()

        cur.execute(
            """
            SELECT *
            FROM blog_article_drafts
            WHERE realtor_id=%s AND article_no=%s
            ORDER BY id DESC
            LIMIT 1
            """,
            (int(realtor_id), str(article_no)),
        )
        return cur.fetchone()


def fetch_article(conn, realtor_id, article_no):
    with conn.cursor() as cur:
        cur.execute(
            """
            SELECT *
            FROM blog_realtor_articles
            WHERE realtor_id=%s AND article_no=%s
            LIMIT 1
            """,
            (int(realtor_id), str(article_no)),
        )
        return cur.fetchone()


def extract_source_json(draft):
    return safe_json_loads(draft.get("source_json")) or {"detail": {}, "images": [], "schools": [], "prices": []}


def count_html_images(html):
    html = html or ""
    return {
        "total": len(re.findall(r"<img\b", html, flags=re.I)),
        "header": len(re.findall(r"realestate-representative-image|header_", html, flags=re.I)),
        "floorplan": len(re.findall(r"평면도|floorplan|photoinfra", html, flags=re.I)),
        "map": len(re.findall(r"realtor_naver_map_|realestate-realtor-map-image", html, flags=re.I)),
        "location": len(re.findall(r"입지|주변환경|land_naver", html, flags=re.I)),
    }


def classify_property_type(value):
    text = clean_text(value).lower()

    if any(k in text for k in ["토지", "대지", "전", "답", "임야"]):
        return "land"
    if any(k in text for k in ["공장", "창고", "지식산업", "제조"]):
        return "factory_warehouse"
    if any(k in text for k in ["상가", "점포", "근린", "상업"]):
        return "store"
    if any(k in text for k in ["사무실", "오피스", "업무"]):
        return "office"
    if any(k in text for k in ["아파트", "apt"]):
        return "apartment"
    if any(k in text for k in ["오피스텔"]):
        return "officetel"
    if any(k in text for k in ["빌라", "연립", "다세대", "주택"]):
        return "villa_house"
    if any(k in text for k in ["분양권", "입주권"]):
        return "presale_right"

    return "unknown"


def classify_trade_type(value):
    text = clean_text(value)

    if "매매" in text:
        return "sale"
    if "전세" in text:
        return "jeonse"
    if "월세" in text:
        return "monthly_rent"
    if "임대" in text:
        return "rent"
    if "분양" in text:
        return "presale"

    return "unknown"


def image_level_for_property(property_class, image_count):
    """
    사진 수 자체로 품질을 낮추지 않는다.
    다만 어떤 레이아웃을 쓸 수 있는지 판단한다.
    """
    if image_count <= 0:
        level = "no_photo"
    elif image_count == 1:
        level = "single_photo"
    elif image_count <= 4:
        level = "compact_photo"
    elif image_count <= 9:
        level = "normal_photo"
    else:
        level = "rich_photo"

    photo_requiredness = "normal"

    if property_class in ["land"]:
        photo_requiredness = "low"
    elif property_class in ["factory_warehouse", "store", "office"]:
        photo_requiredness = "medium_low"
    elif property_class in ["apartment", "officetel", "villa_house"]:
        photo_requiredness = "medium"
    else:
        photo_requiredness = "unknown"

    return {
        "image_level": level,
        "photo_requiredness": photo_requiredness,
        "image_count": image_count,
    }


def data_richness_level(metrics):
    """
    정보량 수준 분류.
    정보량 부족은 FAIL이 아니라 layout 선택 기준이다.
    """
    score = 0
    reasons = []

    if metrics["source_image_count"] >= 10:
        score += 3
        reasons.append("source_image_rich")
    elif metrics["source_image_count"] >= 5:
        score += 2
        reasons.append("source_image_normal")
    elif metrics["source_image_count"] >= 1:
        score += 1
        reasons.append("source_image_low")
    else:
        reasons.append("source_image_none")

    if metrics["broker_desc_len"] >= 250:
        score += 3
        reasons.append("broker_desc_rich")
    elif metrics["broker_desc_len"] >= 80:
        score += 2
        reasons.append("broker_desc_normal")
    elif metrics["broker_desc_len"] >= 20:
        score += 1
        reasons.append("broker_desc_short")
    else:
        reasons.append("broker_desc_none_or_too_short")

    if metrics["has_floorplan"]:
        score += 1
        reasons.append("floorplan_available")

    if metrics["school_count"] > 0:
        score += 1
        reasons.append("school_available")

    if metrics["price_count"] > 0:
        score += 1
        reasons.append("price_available")

    if metrics["has_realtor_map"]:
        score += 1
        reasons.append("realtor_map_available")

    if score >= 8:
        level = "rich"
    elif score >= 5:
        level = "normal"
    elif score >= 3:
        level = "compact"
    else:
        level = "minimal"

    return {
        "data_richness_score": score,
        "data_richness_level": level,
        "reasons": reasons,
    }


def allowed_layout_policy(property_class, trade_class, image_level, richness_level):
    """
    없는 정보를 만들지 않기 위한 layout 정책.
    """
    allowed = set()
    blocked = set()
    notes = []

    # 기본 허용
    allowed.update([
        "representative_image",
        "key_info_table",
        "broker_description",
        "realtor_map",
        "realtor_info",
        "legal_disclosure",
        "seo_footer",
        "simple_notice",
    ])

    if image_level in ["rich_photo", "normal_photo"]:
        allowed.update(["photo_story", "location_photo_story"])
    elif image_level in ["compact_photo", "single_photo"]:
        allowed.update(["single_or_compact_photo"])
        blocked.update(["rich_photo_story", "long_photo_journal"])
        notes.append("사진이 적으므로 사진 중심 장문 레이아웃은 사용하지 않습니다.")
    else:
        blocked.update(["photo_story", "location_photo_story", "rich_photo_story", "long_photo_journal"])
        notes.append("사진이 없으므로 사진 중심 레이아웃을 사용하지 않습니다.")

    if richness_level == "rich":
        allowed.update(["story_intro", "human_intro_from_broker", "detail_review", "sectioned_story"])
    elif richness_level == "normal":
        allowed.update(["human_intro_from_broker", "balanced_review"])
    elif richness_level == "compact":
        allowed.update(["compact_review", "short_human_intro"])
        blocked.update(["long_story_intro", "deep_review"])
        notes.append("정보가 보통 이하이므로 간결한 편집형 글이 적합합니다.")
    else:
        allowed.update(["minimal_listing_note", "fact_based_summary"])
        blocked.update(["long_story_intro", "deep_review", "rich_blog_story", "photo_journal"])
        notes.append("원천 정보가 적으므로 사실 중심의 짧은 글이 적합합니다.")

    if property_class == "land":
        blocked.update(["school_focus", "community_living_story", "apartment_facility_story"])
        allowed.update(["land_fact_summary", "land_use_note"])
        notes.append("토지 매물은 학군/커뮤니티 중심 문장을 사용하지 않습니다.")

    if property_class in ["factory_warehouse"]:
        blocked.update(["school_focus", "family_living_story", "community_living_story"])
        allowed.update(["factory_access_summary", "usage_condition_note"])
        notes.append("공장/창고 매물은 용도, 진입, 면적, 사용 조건 중심으로 작성합니다.")

    if property_class in ["store", "office"]:
        blocked.update(["school_focus", "family_living_story"])
        allowed.update(["business_location_summary", "office_store_condition_note"])
        notes.append("상가/사무실은 입지, 노출, 면적, 임대조건 중심으로 작성합니다.")

    if property_class in ["apartment", "officetel", "villa_house", "presale_right"]:
        allowed.update(["living_condition_summary"])
        # 학군/커뮤니티는 근거가 있을 때만 별도 analyzer에서 활성화

    return {
        "allowed_blocks": sorted(allowed),
        "blocked_blocks": sorted(blocked),
        "policy_notes": notes,
    }


def analyze_available_data(draft, article=None):
    html = draft.get("clipboard_html") or draft.get("draft_html") or ""
    source = extract_source_json(draft)
    detail = source.get("detail") or {}
    source_images = source.get("images") or []
    schools = source.get("schools") or []
    prices = source.get("prices") or []

    raw_json = safe_json_loads(detail.get("raw_json"))
    article_detail = raw_json.get("articleDetail") or {}

    article_name = clean_text(
        detail.get("article_name")
        or draft.get("draft_title")
        or (article or {}).get("article_name")
        or ""
    )

    property_raw = clean_text(
        detail.get("real_estate_type")
        or article_detail.get("realestateTypeName")
        or (article or {}).get("real_estate_type")
        or ""
    )

    trade_raw = clean_text(
        detail.get("trade_type")
        or article_detail.get("tradeTypeName")
        or (article or {}).get("trade_type")
        or ""
    )

    property_class = classify_property_type(property_raw or article_name)
    trade_class = classify_trade_type(trade_raw)

    broker_desc = clean_text(
        detail.get("article_feature_desc")
        or article_detail.get("detailDescription")
        or article_detail.get("articleFeatureDescription")
        or (article or {}).get("article_feature_desc")
        or ""
    )

    html_images = count_html_images(html)

    source_image_count = len(source_images)
    # source_json에 images가 비어 있는 기존 draft도 있으므로 HTML 이미지 기준을 보조로 사용
    if source_image_count <= 0:
        # 대표/지도/배너는 제외하기 어려우므로 보수적으로 total-3 정도를 참고값으로 사용
        source_image_count = max(0, html_images["total"] - html_images["header"] - html_images["map"])

    has_floorplan = html_images["floorplan"] > 0
    has_realtor_map = html_images["map"] > 0

    image_info = image_level_for_property(property_class, source_image_count)

    metrics = {
        "source_image_count": source_image_count,
        "html_image_count": html_images["total"],
        "broker_desc_len": len(broker_desc),
        "has_floorplan": has_floorplan,
        "has_realtor_map": has_realtor_map,
        "school_count": len(schools),
        "price_count": len(prices),
    }

    richness = data_richness_level(metrics)

    policy = allowed_layout_policy(
        property_class=property_class,
        trade_class=trade_class,
        image_level=image_info["image_level"],
        richness_level=richness["data_richness_level"],
    )

    # 작성 모드 결정
    if richness["data_richness_level"] == "rich":
        writing_mode = "rich_editing"
    elif richness["data_richness_level"] == "normal":
        writing_mode = "balanced_editing"
    elif richness["data_richness_level"] == "compact":
        writing_mode = "compact_editing"
    else:
        writing_mode = "minimal_fact_editing"

    confidence_notes = []

    if source_image_count <= 0:
        confidence_notes.append("사진 정보가 없으므로 이미지 설명을 생성하지 않습니다.")
    elif source_image_count == 1:
        confidence_notes.append("사진이 1장이므로 대표 이미지 중심의 간결한 글이 적합합니다.")

    if len(broker_desc) < 20:
        confidence_notes.append("중개사 매물설명이 부족하므로 임의 장점 생성 없이 핵심 정보 중심으로 작성합니다.")

    if property_class == "unknown":
        confidence_notes.append("매물종류가 명확하지 않으므로 범용 사실 중심 레이아웃을 사용합니다.")

    return {
        "draft_id": draft.get("id"),
        "realtor_id": draft.get("realtor_id"),
        "article_no": draft.get("article_no"),
        "title": draft.get("draft_title"),
        "article_name": article_name,
        "property_raw": property_raw,
        "property_class": property_class,
        "trade_raw": trade_raw,
        "trade_class": trade_class,
        "metrics": metrics,
        "image_info": image_info,
        "richness": richness,
        "writing_mode": writing_mode,
        "policy": policy,
        "confidence_notes": confidence_notes,
        "broker_desc_sample": broker_desc[:400],
    }


def print_report(report):
    print("=" * 80)
    print("[STEP111-03 HEE AVAILABLE DATA ANALYZER]")
    print("draft_id:", report.get("draft_id"))
    print("realtor_id:", report.get("realtor_id"))
    print("article_no:", report.get("article_no"))
    print("title:", report.get("title"))
    print("property_raw:", report.get("property_raw"))
    print("property_class:", report.get("property_class"))
    print("trade_raw:", report.get("trade_raw"))
    print("trade_class:", report.get("trade_class"))
    print("-" * 80)
    print("[METRICS]")
    for k, v in report["metrics"].items():
        print(f"{k}: {v}")
    print("-" * 80)
    print("[IMAGE INFO]")
    for k, v in report["image_info"].items():
        print(f"{k}: {v}")
    print("-" * 80)
    print("[RICHNESS]")
    print("score:", report["richness"]["data_richness_score"])
    print("level:", report["richness"]["data_richness_level"])
    print("reasons:", ", ".join(report["richness"]["reasons"]))
    print("writing_mode:", report.get("writing_mode"))
    print("-" * 80)
    print("[ALLOWED BLOCKS]")
    print(", ".join(report["policy"]["allowed_blocks"]))
    print("-" * 80)
    print("[BLOCKED BLOCKS]")
    print(", ".join(report["policy"]["blocked_blocks"]) or "(none)")
    print("-" * 80)
    print("[POLICY NOTES]")
    for note in report["policy"]["policy_notes"]:
        print("-", note)
    if report["confidence_notes"]:
        print("-" * 80)
        print("[CONFIDENCE NOTES]")
        for note in report["confidence_notes"]:
            print("-", note)
    print("-" * 80)
    print("[BROKER DESC SAMPLE]")
    print(report.get("broker_desc_sample") or "(empty)")
    print("=" * 80)


def main():
    parser = argparse.ArgumentParser()
    parser.add_argument("--draft-id", type=int, default=None)
    parser.add_argument("--realtor-id", type=int, default=None)
    parser.add_argument("--article-no", default="")
    parser.add_argument("--json", action="store_true")
    args = parser.parse_args()

    if not args.draft_id and not (args.realtor_id and args.article_no):
        raise RuntimeError("--draft-id 또는 --realtor-id + --article-no 필요")

    conn = get_conn()
    try:
        draft = fetch_draft(conn, draft_id=args.draft_id, realtor_id=args.realtor_id, article_no=args.article_no)
        if not draft:
            raise RuntimeError("draft not found")

        article = fetch_article(conn, draft.get("realtor_id"), draft.get("article_no"))

        report = analyze_available_data(draft, article=article)

        if args.json:
            print(json.dumps(report, ensure_ascii=False, indent=2, default=str))
        else:
            print_report(report)

    finally:
        conn.close()


if __name__ == "__main__":
    main()
