# -*- coding: utf-8 -*-

import os
import re
import json
from bs4 import BeautifulSoup


BASE_DIR = os.path.dirname(os.path.abspath(__file__))
DEBUG_DIR = os.path.join(BASE_DIR, "debug_outputs")


def load_html(article_no):
    path = os.path.join(DEBUG_DIR, f"fin_article_{article_no}.html")

    if not os.path.exists(path):
        print("[ERROR] HTML 파일이 없습니다.")
        print(path)
        print("먼저 debug_fin_article.py를 실행하세요.")
        return ""

    with open(path, "r", encoding="utf-8") as f:
        return f.read()


def clean_text(text):
    text = str(text or "")
    text = re.sub(r"\s+", " ", text)
    return text.strip()


def find_contexts(html, keywords, window=700):
    results = []

    for keyword in keywords:
        if not keyword:
            continue

        for m in re.finditer(re.escape(keyword), html):
            start = max(0, m.start() - window)
            end = min(len(html), m.end() + window)

            results.append({
                "keyword": keyword,
                "position": m.start(),
                "context": html[start:end]
            })

    return results


def find_phone_contexts(html, window=700):
    phone_patterns = [
        r"0\d{1,2}[-.\s]?\d{3,4}[-.\s]?\d{4}",
        r"01\d[-.\s]?\d{3,4}[-.\s]?\d{4}",
        r"\d{8,11}",
    ]

    found = []

    for pattern in phone_patterns:
        for m in re.finditer(pattern, html):
            value = m.group(0)

            start = max(0, m.start() - window)
            end = min(len(html), m.end() + window)

            found.append({
                "phone": value,
                "position": m.start(),
                "context": html[start:end]
            })

    return found


def extract_visible_text(html):
    soup = BeautifulSoup(html, "html.parser")

    for tag in soup(["script", "style", "noscript"]):
        tag.decompose()

    return clean_text(soup.get_text(" "))


def main():
    article_no = input("articleNo 입력: ").strip()

    if not article_no:
        print("articleNo가 없습니다.")
        return

    html = load_html(article_no)

    if not html:
        return

    target_keywords = [
        "044-867-8849",
        "0448678849",
        "대방디엠시티공인중개사사무소",
        "강채원",
        "2624635265",
        "3611011100",
        "중개",
        "공인중개",
        "대표",
        "전화",
        "소재지",
    ]

    keyword_contexts = find_contexts(html, target_keywords, window=900)
    phone_contexts = find_phone_contexts(html, window=900)

    visible_text = extract_visible_text(html)

    visible_contexts = find_contexts(visible_text, target_keywords, window=500)

    report = {
        "article_no": article_no,
        "html_length": len(html),
        "keyword_contexts": keyword_contexts[:100],
        "phone_contexts": phone_contexts[:100],
        "visible_text_length": len(visible_text),
        "visible_contexts": visible_contexts[:100],
    }

    report_path = os.path.join(DEBUG_DIR, f"fin_article_{article_no}_context_report.json")

    with open(report_path, "w", encoding="utf-8") as f:
        json.dump(report, f, ensure_ascii=False, indent=2)

    text_path = os.path.join(DEBUG_DIR, f"fin_article_{article_no}_visible_text.txt")

    with open(text_path, "w", encoding="utf-8") as f:
        f.write(visible_text[:200000])

    print("=" * 80)
    print("[DONE]")
    print("context report:", report_path)
    print("visible text:", text_path)
    print("=" * 80)

    print("[SUMMARY]")
    print("keyword contexts:", len(keyword_contexts))
    print("phone contexts:", len(phone_contexts))
    print("visible contexts:", len(visible_contexts))


if __name__ == "__main__":
    main()