# -*- coding: utf-8 -*-
"""
STEP111-04 HEE Data-Aware Layout Composer v1

목적:
- STEP111-03 Available Data Analyzer 결과를 레이아웃 선택에 반영한다.
- 정보량이 적으면 짧고 정확한 레이아웃을 선택한다.
- 사진이 없으면 사진형 레이아웃을 차단한다.
- 토지/공장/창고/상가/사무실 등 비아파트 매물의 부적절한 블록을 차단한다.
- 운영 초안은 변경하지 않는다. 읽기 전용 추천 도구다.

사용:
  cd /d D:\honghee\blog_api

  draft 기준:
  python tools\hee_data_aware_layout_composer.py --draft-id 209

  article 기준:
  python tools\hee_data_aware_layout_composer.py --realtor-id 1 --article-no 2633019770

  후보 수:
  python tools\hee_data_aware_layout_composer.py --draft-id 209 --count 5

  JSON:
  python tools\hee_data_aware_layout_composer.py --draft-id 209 --json
"""

import argparse
import json
import re
import sys
from pathlib import Path
from html import unescape

ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT))

from db import get_conn


def clean_text(value):
    value = "" if value is None else str(value)
    value = unescape(value)
    value = re.sub(r"<[^>]+>", " ", value)
    value = re.sub(r"\s+", " ", value)
    return value.strip()


def safe_json_loads(value):
    if not value:
        return {}
    if isinstance(value, dict):
        return value
    try:
        return json.loads(value)
    except Exception:
        return {}


def fetch_draft(conn, draft_id=None, realtor_id=None, article_no=None):
    with conn.cursor() as cur:
        if draft_id:
            cur.execute("SELECT * FROM blog_article_drafts WHERE id=%s LIMIT 1", (int(draft_id),))
            return cur.fetchone()

        cur.execute(
            """
            SELECT *
            FROM blog_article_drafts
            WHERE realtor_id=%s AND article_no=%s
            ORDER BY id DESC
            LIMIT 1
            """,
            (int(realtor_id), str(article_no)),
        )
        return cur.fetchone()


def fetch_article(conn, realtor_id, article_no):
    with conn.cursor() as cur:
        cur.execute(
            """
            SELECT *
            FROM blog_realtor_articles
            WHERE realtor_id=%s AND article_no=%s
            LIMIT 1
            """,
            (int(realtor_id), str(article_no)),
        )
        return cur.fetchone()


def extract_source_json(draft):
    return safe_json_loads(draft.get("source_json")) or {"detail": {}, "images": [], "schools": [], "prices": []}


def count_html_images(html):
    html = html or ""
    return {
        "total": len(re.findall(r"<img\b", html, flags=re.I)),
        "header": len(re.findall(r"realestate-representative-image|header_", html, flags=re.I)),
        "floorplan": len(re.findall(r"평면도|floorplan|photoinfra", html, flags=re.I)),
        "map": len(re.findall(r"realtor_naver_map_|realestate-realtor-map-image", html, flags=re.I)),
        "location": len(re.findall(r"입지|주변환경|land_naver", html, flags=re.I)),
    }


def classify_property_type(value):
    text = clean_text(value).lower()

    if any(k in text for k in ["토지", "대지", "전", "답", "임야"]):
        return "land"
    if any(k in text for k in ["공장", "창고", "지식산업", "제조"]):
        return "factory_warehouse"
    if any(k in text for k in ["상가", "점포", "근린", "상업"]):
        return "store"
    if any(k in text for k in ["사무실", "오피스", "업무"]):
        return "office"
    if any(k in text for k in ["아파트", "apt"]):
        return "apartment"
    if any(k in text for k in ["오피스텔"]):
        return "officetel"
    if any(k in text for k in ["빌라", "연립", "다세대", "주택"]):
        return "villa_house"
    if any(k in text for k in ["분양권", "입주권"]):
        return "presale_right"

    return "unknown"


def classify_trade_type(value):
    text = clean_text(value)

    if "매매" in text:
        return "sale"
    if "전세" in text:
        return "jeonse"
    if "월세" in text:
        return "monthly_rent"
    if "임대" in text:
        return "rent"
    if "분양" in text:
        return "presale"

    return "unknown"


def image_level_for_property(property_class, image_count):
    if image_count <= 0:
        level = "no_photo"
    elif image_count == 1:
        level = "single_photo"
    elif image_count <= 4:
        level = "compact_photo"
    elif image_count <= 9:
        level = "normal_photo"
    else:
        level = "rich_photo"

    if property_class == "land":
        requiredness = "low"
    elif property_class in ["factory_warehouse", "store", "office"]:
        requiredness = "medium_low"
    elif property_class in ["apartment", "officetel", "villa_house"]:
        requiredness = "medium"
    else:
        requiredness = "unknown"

    return {
        "image_level": level,
        "photo_requiredness": requiredness,
        "image_count": image_count,
    }


def data_richness_level(metrics):
    score = 0
    reasons = []

    if metrics["source_image_count"] >= 10:
        score += 3
        reasons.append("source_image_rich")
    elif metrics["source_image_count"] >= 5:
        score += 2
        reasons.append("source_image_normal")
    elif metrics["source_image_count"] >= 1:
        score += 1
        reasons.append("source_image_low")
    else:
        reasons.append("source_image_none")

    if metrics["broker_desc_len"] >= 250:
        score += 3
        reasons.append("broker_desc_rich")
    elif metrics["broker_desc_len"] >= 80:
        score += 2
        reasons.append("broker_desc_normal")
    elif metrics["broker_desc_len"] >= 20:
        score += 1
        reasons.append("broker_desc_short")
    else:
        reasons.append("broker_desc_none_or_too_short")

    if metrics["has_floorplan"]:
        score += 1
        reasons.append("floorplan_available")

    if metrics["school_count"] > 0:
        score += 1
        reasons.append("school_available")

    if metrics["price_count"] > 0:
        score += 1
        reasons.append("price_available")

    if metrics["has_realtor_map"]:
        score += 1
        reasons.append("realtor_map_available")

    if score >= 8:
        level = "rich"
    elif score >= 5:
        level = "normal"
    elif score >= 3:
        level = "compact"
    else:
        level = "minimal"

    return {
        "data_richness_score": score,
        "data_richness_level": level,
        "reasons": reasons,
    }


def analyze_available_data(draft, article=None):
    html = draft.get("clipboard_html") or draft.get("draft_html") or ""
    source = extract_source_json(draft)
    detail = source.get("detail") or {}
    source_images = source.get("images") or []
    schools = source.get("schools") or []
    prices = source.get("prices") or []

    raw_json = safe_json_loads(detail.get("raw_json"))
    article_detail = raw_json.get("articleDetail") or {}

    article_name = clean_text(
        detail.get("article_name")
        or draft.get("draft_title")
        or (article or {}).get("article_name")
        or ""
    )

    property_raw = clean_text(
        detail.get("real_estate_type")
        or article_detail.get("realestateTypeName")
        or (article or {}).get("real_estate_type")
        or article_name
        or ""
    )

    trade_raw = clean_text(
        detail.get("trade_type")
        or article_detail.get("tradeTypeName")
        or (article or {}).get("trade_type")
        or ""
    )

    property_class = classify_property_type(property_raw or article_name)
    trade_class = classify_trade_type(trade_raw)

    broker_desc = clean_text(
        detail.get("article_feature_desc")
        or article_detail.get("detailDescription")
        or article_detail.get("articleFeatureDescription")
        or (article or {}).get("article_feature_desc")
        or ""
    )

    html_images = count_html_images(html)

    source_image_count = len(source_images)
    if source_image_count <= 0:
        source_image_count = max(0, html_images["total"] - html_images["header"] - html_images["map"])

    metrics = {
        "source_image_count": source_image_count,
        "html_image_count": html_images["total"],
        "broker_desc_len": len(broker_desc),
        "has_floorplan": html_images["floorplan"] > 0,
        "has_realtor_map": html_images["map"] > 0,
        "school_count": len(schools),
        "price_count": len(prices),
    }

    image_info = image_level_for_property(property_class, source_image_count)
    richness = data_richness_level(metrics)

    if richness["data_richness_level"] == "rich":
        writing_mode = "rich_editing"
    elif richness["data_richness_level"] == "normal":
        writing_mode = "balanced_editing"
    elif richness["data_richness_level"] == "compact":
        writing_mode = "compact_editing"
    else:
        writing_mode = "minimal_fact_editing"

    return {
        "draft_id": draft.get("id"),
        "realtor_id": draft.get("realtor_id"),
        "article_no": draft.get("article_no"),
        "title": draft.get("draft_title"),
        "article_name": article_name,
        "property_raw": property_raw,
        "property_class": property_class,
        "trade_raw": trade_raw,
        "trade_class": trade_class,
        "metrics": metrics,
        "image_info": image_info,
        "richness": richness,
        "writing_mode": writing_mode,
        "broker_desc_sample": broker_desc[:400],
    }


def story_signal_from_data(analysis):
    """
    과장 없이 원천 데이터에서만 story signal을 판단한다.
    정보가 부족하면 generic/minimal로 둔다.
    """
    property_class = analysis["property_class"]
    richness = analysis["richness"]["data_richness_level"]
    metrics = analysis["metrics"]
    text = " ".join([
        analysis.get("title") or "",
        analysis.get("article_name") or "",
        analysis.get("broker_desc_sample") or "",
    ])

    # 매물종류가 우선
    if property_class == "land":
        return "land_fact"
    if property_class == "factory_warehouse":
        return "factory_usage"
    if property_class == "store":
        return "store_business"
    if property_class == "office":
        return "office_business"

    # 정보 적으면 무조건 과한 story 차단
    if richness in ["minimal"]:
        return "minimal_fact"

    if any(k in text for k in ["초", "중", "고", "학교", "학군"]) and property_class in ["apartment", "officetel", "villa_house"]:
        return "school_living"

    if any(k in text for k in ["대단지", "세대", "커뮤니티", "도서관", "게스트하우스", "스포츠센터"]):
        return "complex_living"

    if metrics["has_floorplan"] and property_class in ["apartment", "officetel", "villa_house"]:
        return "structure"

    if analysis["image_info"]["image_level"] in ["rich_photo", "normal_photo"]:
        return "photo_balanced"

    return "balanced_fact"


def base_blocks_by_policy(analysis):
    property_class = analysis["property_class"]
    image_level = analysis["image_info"]["image_level"]
    richness = analysis["richness"]["data_richness_level"]
    metrics = analysis["metrics"]

    required = [
        "representative_image",
        "key_info_table",
        "realtor_info",
        "legal_disclosure",
        "seo_footer",
    ]

    if metrics["has_realtor_map"]:
        required.insert(2, "realtor_map")

    allowed = set(required)
    blocked = set()
    notes = []

    # 설명
    if metrics["broker_desc_len"] >= 20:
        allowed.add("broker_description")
    else:
        blocked.add("long_broker_story")
        notes.append("중개사 설명이 짧아 장문 설명을 만들지 않습니다.")

    # 사진
    if image_level in ["rich_photo", "normal_photo"]:
        allowed.update(["photo_story", "location_photo_story"])
    elif image_level in ["compact_photo", "single_photo"]:
        allowed.add("single_or_compact_photo")
        blocked.update(["photo_story", "location_photo_story", "rich_photo_story"])
        notes.append("사진 수가 적어 사진 중심 레이아웃을 제한합니다.")
    else:
        blocked.update(["single_or_compact_photo", "photo_story", "location_photo_story", "rich_photo_story"])
        notes.append("사진이 없으므로 사진 블록을 사용하지 않습니다.")

    # 평면도
    if metrics["has_floorplan"]:
        allowed.add("floorplan")

    # 정보량별 글 길이
    if richness == "rich":
        allowed.update(["human_intro", "sectioned_story", "detail_review", "checklist"])
    elif richness == "normal":
        allowed.update(["human_intro", "balanced_note", "checklist"])
    elif richness == "compact":
        allowed.update(["short_intro", "compact_note"])
        blocked.update(["sectioned_story", "detail_review", "long_story"])
        notes.append("정보량이 compact 수준이므로 짧고 정확한 구성만 사용합니다.")
    else:
        allowed.update(["minimal_fact_note"])
        blocked.update(["sectioned_story", "detail_review", "long_story", "checklist"])
        notes.append("정보량이 minimal 수준이므로 사실 중심 요약만 사용합니다.")

    # 매물 종류별 차단
    if property_class == "land":
        allowed.update(["land_fact_note", "land_condition_table"])
        blocked.update(["school_living_note", "community_note", "apartment_living_story", "floorplan"])
        notes.append("토지 매물은 학군/커뮤니티/평면도 중심 구성을 사용하지 않습니다.")
    elif property_class == "factory_warehouse":
        allowed.update(["factory_usage_note", "access_condition_note"])
        blocked.update(["school_living_note", "community_note", "family_living_story"])
        notes.append("공장/창고는 용도, 진입, 면적, 사용 조건 중심으로 구성합니다.")
    elif property_class == "store":
        allowed.update(["business_location_note", "store_condition_note"])
        blocked.update(["school_living_note", "family_living_story"])
        notes.append("상가는 노출, 입지, 면적, 임대/매매 조건 중심으로 구성합니다.")
    elif property_class == "office":
        allowed.update(["office_location_note", "office_condition_note"])
        blocked.update(["school_living_note", "family_living_story"])
        notes.append("사무실은 업무 접근성, 면적, 관리 조건 중심으로 구성합니다.")
    elif property_class in ["apartment", "officetel", "villa_house"]:
        allowed.add("living_condition_summary")
        if metrics["school_count"] > 0 or "학교" in (analysis.get("broker_desc_sample") or ""):
            allowed.add("school_living_note")
        if any(k in (analysis.get("broker_desc_sample") or "") for k in ["커뮤니티", "도서관", "게스트하우스", "스포츠센터"]):
            allowed.add("community_note")

    return {
        "required_blocks": required,
        "allowed_blocks": sorted(allowed),
        "blocked_blocks": sorted(blocked),
        "policy_notes": notes,
    }


def compose_layout_candidates(analysis, count=3):
    policy = base_blocks_by_policy(analysis)
    allowed = set(policy["allowed_blocks"])
    blocked = set(policy["blocked_blocks"])
    story_signal = story_signal_from_data(analysis)

    flows = []

    # 최소형
    if analysis["richness"]["data_richness_level"] == "minimal":
        flows.append([
            "representative_image",
            "key_info_table",
            "minimal_fact_note",
            "broker_description",
            "realtor_map",
            "realtor_info",
            "legal_disclosure",
            "seo_footer",
        ])

    # compact
    if analysis["richness"]["data_richness_level"] == "compact":
        flows.extend([
            ["representative_image", "short_intro", "key_info_table", "single_or_compact_photo", "broker_description", "realtor_map", "realtor_info", "legal_disclosure", "seo_footer"],
            ["representative_image", "key_info_table", "compact_note", "broker_description", "single_or_compact_photo", "realtor_map", "realtor_info", "legal_disclosure", "seo_footer"],
        ])

    # normal/rich
    if analysis["richness"]["data_richness_level"] in ["normal", "rich"]:
        if story_signal in ["school_living", "complex_living"]:
            flows.extend([
                ["representative_image", "human_intro", "school_living_note", "community_note", "location_photo_story", "broker_description", "key_info_table", "floorplan", "photo_story", "realtor_map", "realtor_info", "legal_disclosure", "seo_footer"],
                ["representative_image", "human_intro", "broker_description", "photo_story", "key_info_table", "floorplan", "school_living_note", "community_note", "realtor_map", "realtor_info", "legal_disclosure", "seo_footer"],
                ["representative_image", "photo_story", "human_intro", "key_info_table", "broker_description", "floorplan", "school_living_note", "realtor_map", "realtor_info", "legal_disclosure", "seo_footer"],
            ])
        elif story_signal == "structure":
            flows.extend([
                ["representative_image", "human_intro", "key_info_table", "floorplan", "balanced_note", "photo_story", "broker_description", "checklist", "realtor_map", "realtor_info", "legal_disclosure", "seo_footer"],
                ["representative_image", "photo_story", "human_intro", "floorplan", "key_info_table", "broker_description", "realtor_map", "realtor_info", "legal_disclosure", "seo_footer"],
            ])
        elif story_signal in ["photo_balanced"]:
            flows.extend([
                ["representative_image", "human_intro", "photo_story", "key_info_table", "location_photo_story", "broker_description", "floorplan", "realtor_map", "realtor_info", "legal_disclosure", "seo_footer"],
                ["representative_image", "photo_story", "human_intro", "key_info_table", "broker_description", "floorplan", "realtor_map", "realtor_info", "legal_disclosure", "seo_footer"],
            ])
        else:
            flows.extend([
                ["representative_image", "human_intro", "key_info_table", "broker_description", "floorplan", "photo_story", "balanced_note", "realtor_map", "realtor_info", "legal_disclosure", "seo_footer"],
                ["representative_image", "key_info_table", "balanced_note", "broker_description", "single_or_compact_photo", "realtor_map", "realtor_info", "legal_disclosure", "seo_footer"],
            ])

    # 비주거/토지 전용
    if story_signal == "land_fact":
        flows = [
            ["representative_image", "key_info_table", "land_fact_note", "land_condition_table", "broker_description", "realtor_map", "realtor_info", "legal_disclosure", "seo_footer"],
            ["representative_image", "short_intro", "key_info_table", "broker_description", "land_fact_note", "realtor_map", "realtor_info", "legal_disclosure", "seo_footer"],
        ]
    elif story_signal == "factory_usage":
        flows = [
            ["representative_image", "key_info_table", "factory_usage_note", "access_condition_note", "single_or_compact_photo", "broker_description", "realtor_map", "realtor_info", "legal_disclosure", "seo_footer"],
        ]
    elif story_signal == "store_business":
        flows = [
            ["representative_image", "key_info_table", "business_location_note", "store_condition_note", "single_or_compact_photo", "broker_description", "realtor_map", "realtor_info", "legal_disclosure", "seo_footer"],
        ]
    elif story_signal == "office_business":
        flows = [
            ["representative_image", "key_info_table", "office_location_note", "office_condition_note", "single_or_compact_photo", "broker_description", "realtor_map", "realtor_info", "legal_disclosure", "seo_footer"],
        ]

    candidates = []
    for idx, flow in enumerate(flows):
        cleaned = []
        seen = set()

        for block in flow:
            if block in blocked:
                continue
            if block not in allowed:
                continue
            if block in seen:
                continue
            cleaned.append(block)
            seen.add(block)

        # 필수 보강
        for block in policy["required_blocks"]:
            if block not in cleaned and block in allowed:
                cleaned.append(block)

        score = score_layout(cleaned, analysis, policy, story_signal)

        candidates.append({
            "layout_id": f"{analysis['writing_mode']}_{story_signal}_v{idx+1:02d}",
            "story_signal": story_signal,
            "blocks": cleaned,
            "score": score["score"],
            "score_detail": score,
        })

    # 중복 제거
    unique = []
    for c in candidates:
        if c["blocks"] not in [u["blocks"] for u in unique]:
            unique.append(c)

    unique = sorted(unique, key=lambda x: x["score"], reverse=True)

    return {
        "analysis": analysis,
        "policy": policy,
        "story_signal": story_signal,
        "candidates": unique[:count],
    }


def score_layout(blocks, analysis, policy, story_signal):
    score = 100
    penalties = []
    bonuses = []

    for rb in policy["required_blocks"]:
        if rb not in blocks:
            score -= 40
            penalties.append(f"required_missing:{rb}")

    # 없는 정보 기반 블록 있으면 큰 감점
    for b in blocks:
        if b in policy["blocked_blocks"]:
            score -= 30
            penalties.append(f"blocked_block_used:{b}")

    richness = analysis["richness"]["data_richness_level"]
    image_level = analysis["image_info"]["image_level"]

    # 사진 없는/적은 매물에서 사진형 흐름 방지
    if image_level in ["no_photo", "single_photo"] and any(b in blocks for b in ["photo_story", "location_photo_story"]):
        score -= 30
        penalties.append("photo_story_not_allowed_for_low_image")

    # minimal인데 장문형 블록 있으면 감점
    if richness == "minimal" and any(b in blocks for b in ["sectioned_story", "detail_review", "long_story", "photo_story"]):
        score -= 25
        penalties.append("too_rich_for_minimal_data")

    # 도입부 자연성
    if any(b in blocks[:3] for b in ["human_intro", "short_intro", "minimal_fact_note"]):
        score += 5
        bonuses.append("natural_intro_early")

    # 표가 너무 앞이어도 minimal/land/factory/store는 허용
    if "key_info_table" in blocks and blocks.index("key_info_table") <= 1:
        if story_signal not in ["land_fact", "factory_usage", "store_business", "office_business", "minimal_fact"]:
            score -= 4
            penalties.append("key_info_early_but_acceptable")

    # 법정표시는 후반
    if "legal_disclosure" in blocks and blocks.index("legal_disclosure") < max(3, len(blocks) - 3):
        score -= 4
        penalties.append("legal_too_early")

    # 지도는 후반
    if "realtor_map" in blocks and blocks.index("realtor_map") < 5 and len(blocks) > 7:
        score -= 4
        penalties.append("map_too_early")

    score = max(0, min(120, score))

    return {
        "score": score,
        "bonuses": bonuses,
        "penalties": penalties,
    }


def print_report(result):
    analysis = result["analysis"]
    policy = result["policy"]

    print("=" * 80)
    print("[STEP111-04 HEE DATA-AWARE LAYOUT COMPOSER]")
    print("draft_id:", analysis.get("draft_id"))
    print("realtor_id:", analysis.get("realtor_id"))
    print("article_no:", analysis.get("article_no"))
    print("title:", analysis.get("title"))
    print("property_class:", analysis.get("property_class"), f"({analysis.get('property_raw')})")
    print("trade_class:", analysis.get("trade_class"), f"({analysis.get('trade_raw')})")
    print("image_level:", analysis["image_info"]["image_level"], "count=", analysis["image_info"]["image_count"])
    print("richness:", analysis["richness"]["data_richness_level"], "score=", analysis["richness"]["data_richness_score"])
    print("writing_mode:", analysis.get("writing_mode"))
    print("story_signal:", result.get("story_signal"))
    print("-" * 80)
    print("[POLICY NOTES]")
    for note in policy.get("policy_notes") or []:
        print("-", note)
    if not policy.get("policy_notes"):
        print("(none)")
    print("-" * 80)
    print("[BLOCKED BLOCKS]")
    print(", ".join(policy.get("blocked_blocks") or []) or "(none)")
    print("-" * 80)

    for idx, cand in enumerate(result["candidates"], start=1):
        print(f"[CANDIDATE {idx}] {cand['layout_id']} score={cand['score']}")
        if cand["score_detail"].get("bonuses"):
            print(" bonuses:", ", ".join(cand["score_detail"]["bonuses"]))
        if cand["score_detail"].get("penalties"):
            print(" penalties:", ", ".join(cand["score_detail"]["penalties"]))
        print(" blocks:")
        for i, b in enumerate(cand["blocks"], start=1):
            print(f"  {i:02d}. {b}")
        print("-" * 80)

    print("[BROKER DESC SAMPLE]")
    print(analysis.get("broker_desc_sample") or "(empty)")
    print("=" * 80)


def main():
    parser = argparse.ArgumentParser()
    parser.add_argument("--draft-id", type=int, default=None)
    parser.add_argument("--realtor-id", type=int, default=None)
    parser.add_argument("--article-no", default="")
    parser.add_argument("--count", type=int, default=3)
    parser.add_argument("--json", action="store_true")
    args = parser.parse_args()

    if not args.draft_id and not (args.realtor_id and args.article_no):
        raise RuntimeError("--draft-id 또는 --realtor-id + --article-no 필요")

    conn = get_conn()
    try:
        draft = fetch_draft(conn, draft_id=args.draft_id, realtor_id=args.realtor_id, article_no=args.article_no)
        if not draft:
            raise RuntimeError("draft not found")

        article = fetch_article(conn, draft.get("realtor_id"), draft.get("article_no"))
        analysis = analyze_available_data(draft, article=article)
        result = compose_layout_candidates(analysis, count=args.count)

        if args.json:
            print(json.dumps(result, ensure_ascii=False, indent=2, default=str))
        else:
            print_report(result)

    finally:
        conn.close()


if __name__ == "__main__":
    main()
