import os
import sys
import json
import time
import glob
import logging
import threading
import random
from concurrent.futures import ThreadPoolExecutor, as_completed

SCRIPTS_DIR = os.path.dirname(os.path.abspath(__file__))
ROOT_DIR = os.path.dirname(SCRIPTS_DIR)

if ROOT_DIR not in sys.path:
    sys.path.insert(0, ROOT_DIR)

from pinscrape.v2 import Pinterest, fetch_related_keywords
from pinscrape.database import get_pending_snippet_keywords, update_snippet
from pinscrape.utils import load_proxies, format_description

logging.basicConfig(
    level=logging.INFO,
    format="%(asctime)s [%(levelname)s] %(message)s",
    datefmt="%H:%M:%S"
)
logger = logging.getLogger(__name__)


def load_config():
    config_path = os.path.join(ROOT_DIR, "config.json")
    defaults = {
        "image_result": 26,
        "min_snippet": 3,
        "concurrency": 5,
        "max_retries": 3,
        "retry_delay": 3
    }
    if os.path.exists(config_path):
        with open(config_path, "r") as f:
            cfg = json.load(f)
        defaults.update(cfg)
    return defaults


def scrape_keyword_description(db_path, keyword_id, keyword, proxy_list, config):
    base_page_size = config["image_result"]
    min_snippet = config.get("min_snippet", 3)
    max_page_size = config.get("snippet_max_page_size", 50)
    max_network_retries = config["max_retries"]
    hl = config.get("suggest_hl", "en")
    gl = config.get("suggest_gl", "US")
    max_related = config.get("max_related_kw", 10)

    related_kw = fetch_related_keywords(keyword, proxy_list=proxy_list, max_results=max_related, hl=hl, gl=gl)

    p = Pinterest(proxy_list=proxy_list, sleep_time=(1.5, 4.0), max_retries=3)

    best_descriptions = []
    current_page_size = base_page_size

    while current_page_size <= max_page_size:
        for attempt in range(1, max_network_retries + 1):
            try:
                descriptions = p.search_descriptions(keyword, page_size=current_page_size)
                if descriptions and len(descriptions) > len(best_descriptions):
                    best_descriptions = descriptions

                break
            except Exception as e:
                logger.warning(f"  Attempt {attempt}/{max_network_retries} failed for '{keyword}' (page_size={current_page_size}): {e}")
                if attempt < max_network_retries:
                    wait = random.uniform(3, 8)
                    logger.info(f"  Waiting {wait:.1f}s before retry...")
                    time.sleep(wait)
                    p.reset_session()

        if len(best_descriptions) >= min_snippet:
            break
        current_page_size += 10

    cleaned = [format_description(d) for d in best_descriptions if format_description(d)]

    if not cleaned and not related_kw:
        return {
            "id": keyword_id,
            "keyword": keyword,
            "desc_count": 0,
            "related_count": 0,
            "error": "no data found (skipped save)"
        }

    snippet_data = {
        "related_kw": related_kw,
        "description": cleaned
    }
    snippet_json = json.dumps(snippet_data, ensure_ascii=False)
    update_snippet(db_path, keyword_id, snippet_json)

    return {
        "id": keyword_id,
        "keyword": keyword,
        "desc_count": len(cleaned),
        "related_count": len(related_kw),
        "error": None
    }


def scrape_database_descriptions(db_path, proxy_list, config):
    db_name = os.path.basename(db_path)
    logger.info(f"\n{'='*50}")
    logger.info(f"Processing: {db_name}")
    logger.info(f"{'='*50}")

    pending = get_pending_snippet_keywords(db_path)
    if not pending:
        logger.info(f"No pending keywords for snippet in {db_name}")
        return

    logger.info(f"Found {len(pending)} keywords without snippet")
    concurrency = config["concurrency"]
    completed = 0
    lock = threading.Lock()

    def progress_callback(future):
        nonlocal completed
        result = future.result()
        with lock:
            completed += 1
            status = "OK" if not result["error"] else f"FAIL ({result['error']})"
            logger.info(f"  [{completed}/{len(pending)}] {result['keyword']} -> {result['desc_count']} desc, {result['related_count']} related [{status}]")

    with ThreadPoolExecutor(max_workers=concurrency) as executor:
        futures = {}
        for item in pending:
            future = executor.submit(
                scrape_keyword_description,
                db_path, item["id"], item["keyword"],
                proxy_list, config
            )
            futures[future] = item
            future.add_done_callback(progress_callback)

        for future in as_completed(futures):
            pass

    logger.info(f"Completed {db_name}: {completed}/{len(pending)} keywords processed")


def find_databases():
    db_dir = os.path.join(ROOT_DIR, "database")
    if not os.path.exists(db_dir):
        return []
    return sorted(glob.glob(os.path.join(db_dir, "*.sqlite")))


def main():
    config = load_config()
    logger.info(f"Config: image_result={config['image_result']}, min_snippet={config.get('min_snippet', 3)}, "
                f"snippet_max_page_size={config.get('snippet_max_page_size', 50)}, "
                f"max_related_kw={config.get('max_related_kw', 10)}, "
                f"suggest_hl={config.get('suggest_hl', 'en')}, suggest_gl={config.get('suggest_gl', 'US')}, "
                f"concurrency={config['concurrency']}, max_retries={config['max_retries']}, retry_delay={config['retry_delay']}s")

    proxy_list = load_proxies(os.path.join(ROOT_DIR, "proxies.txt"))
    logger.info(f"Loaded {len(proxy_list)} proxies")

    databases = find_databases()
    if not databases:
        logger.warning("No .sqlite files found in database/ folder")
        return

    logger.info(f"Found {len(databases)} database(s): {[os.path.basename(d) for d in databases]}")

    for db_path in databases:
        scrape_database_descriptions(db_path, proxy_list, config)

    logger.info("\nAll databases processed!")


if __name__ == "__main__":
    main()
