import csv
import glob
import json
import os
import secrets
import ssl
import string
import subprocess
import sys
import time
import urllib.request
import urllib.error
import xml.etree.ElementTree as ET
from datetime import datetime, timezone
from urllib.parse import urlparse, urlunparse, quote

def _build_ssl_ctx():
    ctx = ssl.create_default_context()
    try:
        import certifi
        ctx.load_verify_locations(certifi.where())
        return ctx
    except Exception:
        pass
    if os.name == "nt":
        try:
            certs = ssl.enum_certificates("root")
            import tempfile
            cafile = os.path.join(tempfile.gettempdir(), "_indexnow_ca.pem")
            with open(cafile, "w") as f:
                for der, _, _ in certs:
                    f.write(ssl.DER_cert_to_PEM_cert(der))
            ctx.load_verify_locations(cafile)
            return ctx
        except Exception:
            pass
    ctx.check_hostname = False
    ctx.verify_mode = ssl.CERT_NONE
    return ctx

SSL_CTX = _build_ssl_ctx()

SCRIPTS_DIR = os.path.dirname(os.path.abspath(__file__))
ROOT_DIR = os.path.dirname(SCRIPTS_DIR)

CONFIG = {
    'create_csv': os.path.join(ROOT_DIR, 'CREATE.csv'),
    'export_folder': os.path.join(ROOT_DIR, 'export'),
    'report_dir': os.path.join(ROOT_DIR, 'submit/Report PY'),
    'summary_dir': os.path.join(ROOT_DIR, 'submit/Summary Report'),
    'key_dir': os.path.join(ROOT_DIR, 'submit/key'),
    'report_folder': os.path.join(ROOT_DIR, 'report'),
    'indexnow_api': 'https://searchadvisor.naver.com/indexnow',
    'max_retries': 3,
    'max_urls_per_batch': 10000,
    'host': 'storage.googleapis.com',
}

HEADERS = {"Content-Type": "application/json; charset=utf-8"}
GCLOUD_CMD = None


def _resolve_gcloud():
    candidates = [
        os.path.join(os.environ.get("LOCALAPPDATA", ""),
                     "Google", "Cloud SDK", "google-cloud-sdk", "bin", "gcloud.cmd"),
        os.path.join(os.environ.get("ProgramFiles", ""),
                     "Google", "Cloud SDK", "google-cloud-sdk", "bin", "gcloud.cmd"),
        os.path.join(os.environ.get("ProgramFiles(x86)", ""),
                     "Google", "Cloud SDK", "google-cloud-sdk", "bin", "gcloud.cmd"),
    ]
    for p in candidates:
        if os.path.isfile(p):
            return p
    return "gcloud"


def generate_key(length=36):
    chars = string.ascii_lowercase + string.digits
    return "".join(secrets.choice(chars) for _ in range(length))


def log(report_dir, bucket, message):
    os.makedirs(report_dir, exist_ok=True)
    log_file = os.path.join(report_dir, f"{bucket}_indexnow_log.txt")
    timestamp = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
    line = f"[{timestamp}] {message}"
    print(f"  {line}")
    with open(log_file, "a", encoding="utf-8") as f:
        f.write(line + "\n")


def fetch_url(url):
    req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0"})
    with urllib.request.urlopen(req, timeout=30, context=SSL_CTX) as resp:
        return resp.read()


def extract_sitemap_urls(content):
    urls = []
    root = ET.fromstring(content)
    ns_map = {"ns": "http://www.sitemaps.org/schemas/sitemap/0.9"}
    if root.tag.endswith("sitemapindex"):
        for sm in root.findall("ns:sitemap", ns_map):
            loc = sm.find("ns:loc", ns_map)
            if loc is not None and loc.text:
                urls.append(loc.text.strip())
    elif root.tag.endswith("urlset"):
        for u in root.findall("ns:url", ns_map):
            loc = u.find("ns:loc", ns_map)
            if loc is not None and loc.text:
                urls.append(loc.text.strip())
    return urls


def load_submitted(report_dir, bucket):
    path = os.path.join(report_dir, f"{bucket}_submitted_urls.txt")
    if not os.path.exists(path):
        return set()
    with open(path, "r", encoding="utf-8") as f:
        return set(line.strip() for line in f if line.strip())


def save_submitted(report_dir, bucket, urls):
    path = os.path.join(report_dir, f"{bucket}_submitted_urls.txt")
    with open(path, "a", encoding="utf-8") as f:
        for url in urls:
            f.write(url + "\n")


def submit_indexnow(host, key, key_location, url_list):
    payload = json.dumps({
        "host": host,
        "key": key,
        "keyLocation": key_location,
        "urlList": url_list
    }, ensure_ascii=False).encode("utf-8")

    for attempt in range(1, CONFIG['max_retries'] + 1):
        req = urllib.request.Request(
            CONFIG['indexnow_api'],
            data=payload,
            headers=HEADERS,
            method="POST"
        )
        try:
            with urllib.request.urlopen(req, timeout=30, context=SSL_CTX) as resp:
                code = resp.getcode()
            return code, None
        except urllib.error.HTTPError as e:
            body = e.read().decode("utf-8", errors="replace")
            if e.code == 403 and attempt < CONFIG['max_retries']:
                time.sleep(10)
                continue
            if "SiteVerificationNotCompleted" in body and attempt < CONFIG['max_retries']:
                time.sleep(10)
                continue
            return e.code, body
        except urllib.error.URLError as e:
            if attempt < CONFIG['max_retries']:
                time.sleep(10)
                continue
            return 0, str(e.reason)

    return 0, "Max retries exceeded"


def upload_key_to_gcs(local_path, bucket):
    global GCLOUD_CMD
    dest = f"gs://{bucket}/"
    abs_path = os.path.abspath(local_path)
    try:
        result = subprocess.run(
            [GCLOUD_CMD, "storage", "cp", abs_path, dest, "--quiet"],
            capture_output=True, text=True, timeout=60
        )
        return result.returncode == 0, result.stderr.strip()
    except subprocess.TimeoutExpired:
        return False, "gcloud command timed out"
    except FileNotFoundError:
        return False, "gcloud CLI not found"
    except Exception as e:
        return False, str(e)


def read_create_csv(filepath):
    mappings = []
    with open(filepath, 'r', encoding='utf-8-sig') as f:
        reader = csv.DictReader(f)
        for row in reader:
            mappings.append(row)
    return mappings


def count_html_files(folder_path):
    return len(glob.glob(os.path.join(folder_path, '*.html')))


def generate_summary_json(results, report_folder):
    if not os.path.exists(report_folder):
        os.makedirs(report_folder)

    summary = []
    for r in results:
        summary.append({
            'bucket': r['bucket'],
            'url_count': r['url_count'],
            'submit': r['status'] if r['status'] in ('SUCCESS', 'SKIP') else 'FAILED',
            'notes': r.get('notes', ''),
            'source': r.get('source', ''),
        })

    json_file = os.path.join(report_folder, 'submit-summary.json')
    with open(json_file, 'w', encoding='utf-8') as f:
        json.dump(summary, f, indent=2)

    return json_file


def main():
    global GCLOUD_CMD
    GCLOUD_CMD = _resolve_gcloud()

    import argparse
    parser = argparse.ArgumentParser(description='IndexNow Submission from CREATE.csv')
    parser.add_argument('--create-csv', default=CONFIG['create_csv'], help='Path to CREATE.csv')
    parser.add_argument('--export-folder', default=CONFIG['export_folder'], help='Export folder path')
    args = parser.parse_args()

    if not os.path.exists(args.create_csv):
        print(f"Error: {args.create_csv} not found.")
        sys.exit(1)

    os.makedirs(CONFIG['report_dir'], exist_ok=True)
    os.makedirs(CONFIG['key_dir'], exist_ok=True)

    mappings = read_create_csv(args.create_csv)

    if not mappings:
        print("Error: No buckets found in CREATE.csv.")
        sys.exit(0)

    print("=" * 60)
    print("  IndexNow Submission (CREATE.csv)")
    print("=" * 60)
    print(f"Found {len(mappings)} buckets to process")
    print()

    results = []
    total_urls_submitted = 0
    success_count = 0
    fail_count = 0
    skip_count = 0

    for idx, m in enumerate(mappings, 1):
        bucket = m['bucket']
        foldering = m.get('foldering', '')

        print(f"\n{'=' * 50}")
        print(f"[{idx}/{len(mappings)}] Processing bucket: {bucket}")
        print(f"{'=' * 50}")

        entry = {"bucket": bucket, "key": "", "url_count": 0, "status": "FAIL", "notes": "", "source": m.get('source', '')}

        folder_path = os.path.join(args.export_folder, foldering)
        if not foldering or not os.path.isdir(folder_path):
            msg = f"Export folder not found: {foldering}"
            log(CONFIG['report_dir'], bucket, msg)
            entry["notes"] = msg
            entry["status"] = "SKIP"
            results.append(entry)
            skip_count += 1
            continue

        key = generate_key()
        entry["key"] = key
        key_file = os.path.join(CONFIG['key_dir'], f"{key}.txt")

        with open(key_file, "w", encoding="utf-8") as f:
            f.write(key)

        log(CONFIG['report_dir'], bucket, f"Key generated: {key}")

        log(CONFIG['report_dir'], bucket, "Uploading key to gcs...")
        upload_ok, err_msg = upload_key_to_gcs(key_file, bucket)

        if not upload_ok:
            msg = f"GCS upload failed: {err_msg}"
            log(CONFIG['report_dir'], bucket, msg)
            entry["notes"] = msg
            results.append(entry)
            fail_count += 1
            continue

        log(CONFIG['report_dir'], bucket, "Key uploaded.")
        key_location = f"https://storage.googleapis.com/{bucket}/{key}.txt"

        sitemap_url = f"https://storage.googleapis.com/{bucket}/sitemap.xml"
        log(CONFIG['report_dir'], bucket, f"Fetching sitemap: {sitemap_url}")

        try:
            content = fetch_url(sitemap_url)
        except Exception as e:
            msg = f"Sitemap not found: {e}"
            log(CONFIG['report_dir'], bucket, msg)
            entry["notes"] = msg
            results.append(entry)
            fail_count += 1
            continue

        all_urls = []
        try:
            parsed = extract_sitemap_urls(content)
            if b"<sitemapindex" in content:
                subsitemaps = parsed
                log(CONFIG['report_dir'], bucket, f"Found {len(subsitemaps)} subsitemaps.")
                for sub_url in subsitemaps:
                    try:
                        sub_content = fetch_url(sub_url)
                        all_urls.extend(extract_sitemap_urls(sub_content))
                    except Exception as e:
                        log(CONFIG['report_dir'], bucket, f"Sub-sitemap failed: {sub_url} - {e}")
            else:
                all_urls = parsed
        except ET.ParseError as e:
            log(CONFIG['report_dir'], bucket, f"XML parse error: {e}")

        all_urls = sorted(set(u.strip() for u in all_urls if u.strip()))

        all_file = os.path.join(CONFIG['report_dir'], f"{bucket}_urls_all.txt")
        with open(all_file, "w", encoding="utf-8") as f:
            for u in all_urls:
                f.write(u + "\n")

        if not all_urls:
            msg = "No URLs found in sitemap."
            log(CONFIG['report_dir'], bucket, msg)
            entry["notes"] = msg

        submitted = load_submitted(CONFIG['report_dir'], bucket)
        new_urls_raw = [u for u in all_urls if u not in submitted]

        if not new_urls_raw and all_urls:
            msg = f"All {len(all_urls)} URLs previously submitted."
            log(CONFIG['report_dir'], bucket, msg)
            entry["url_count"] = len(all_urls)
            entry["status"] = "SKIP"
            entry["notes"] = msg
            results.append(entry)
            success_count += 1
            continue

        to_submit_raw = new_urls_raw[:CONFIG['max_urls_per_batch']]
        pending_raw = new_urls_raw[CONFIG['max_urls_per_batch']:]

        to_submit = []
        for u in to_submit_raw:
            parts = urlparse(u)
            path = quote(parts.path, safe="/")
            query = quote(parts.query, safe="=&")
            to_submit.append(urlunparse((parts.scheme, parts.netloc, path, parts.params, query, parts.fragment)))

        submit_file = os.path.join(CONFIG['report_dir'], f"{bucket}_urls_to_submit.txt")
        with open(submit_file, "w", encoding="utf-8") as f:
            for u in to_submit:
                f.write(u + "\n")

        payload_file = os.path.join(CONFIG['report_dir'], f"{bucket}_payload.json")
        payload = {"host": CONFIG['host'], "key": key, "keyLocation": key_location, "urlList": to_submit}
        with open(payload_file, "w", encoding="utf-8") as f:
            json.dump(payload, f, indent=2)

        log(CONFIG['report_dir'], bucket, f"Submitting {len(to_submit)} URLs...")
        code, err_body = submit_indexnow(CONFIG['host'], key, key_location, to_submit)

        if code == 200:
            entry["url_count"] = len(to_submit)
            entry["status"] = "SUCCESS"
            entry["notes"] = "HTTP 200"
            log(CONFIG['report_dir'], bucket, f"SUCCESS: {len(to_submit)} URLs submitted.")
            save_submitted(CONFIG['report_dir'], bucket, to_submit_raw)
            total_urls_submitted += len(to_submit)
            success_count += 1
        else:
            entry["url_count"] = len(to_submit)
            entry["status"] = "FAIL"
            entry["notes"] = f"HTTP {code}: {err_body or ''}"
            log(CONFIG['report_dir'], bucket, f"FAILED: HTTP {code} {err_body or ''}")
            fail_count += 1

        if pending_raw:
            pending_file = os.path.join(CONFIG['report_dir'], f"{bucket}_pending_urls.txt")
            with open(pending_file, "w", encoding="utf-8") as f:
                for u in pending_raw:
                    f.write(u + "\n")
            log(CONFIG['report_dir'], bucket, f"{len(pending_raw)} URLs pending (exceeds 10k).")

        results.append(entry)

    print(f"\n{'=' * 70}")
    print("INDEXNOW SUBMISSION REPORT")
    print(f"Date: {datetime.now().strftime('%Y-%m-%d %H:%M:%S')}")
    print(f"{'=' * 70}")
    print(f"{'No':>3} | {'Bucket Name':32s} | {'Key (partial)':15s} | {'URLs':>5} | {'Status':7s} | Notes")
    print("-" * 85)
    for i, r in enumerate(results, 1):
        k = r["key"][:12] + "..." if len(r["key"]) > 12 else r["key"]
        n = (r.get("notes") or "")[:35]
        print(f"{i:3d} | {r['bucket']:32s} | {k:15s} | {r['url_count']:5d} | {r['status']:7s} | {n}")
    print("-" * 85)
    print(f"Total: {len(results)} processed, {success_count} success, {skip_count} skipped, {fail_count} failed")
    print(f"Total URLs submitted: {total_urls_submitted}")
    print(f"{'=' * 70}")

    generate_summary_json(results, CONFIG['report_folder'])


if __name__ == '__main__':
    main()
