#!/usr/bin/env python3
"""Scarica in sequenza le immagini indicate nei documenti MongoDB."""

from __future__ import annotations

import argparse
from concurrent.futures import FIRST_COMPLETED, Future, ThreadPoolExecutor, wait
import mimetypes
import re
import shutil
import sys
import threading
import unicodedata
from dataclasses import dataclass
from pathlib import Path
from collections.abc import Iterable, Iterator, Sequence
from urllib.error import HTTPError, URLError
from urllib.parse import unquote, urlparse
from urllib.request import Request, urlopen

from config import (
    DOWNLOAD_WORKERS,
    DOCUMENT_NAME_FIELD,
    DOCUMENT_SLUG_FIELD,
    IMAGES_FIELD,
    MAX_PENDING_DOWNLOADS,
    MONGODB_BATCH_SIZE,
    MONGODB_COLLECTION,
    MONGODB_DATABASE,
    MONGODB_TIMEOUT_MS,
    MONGODB_URI,
)


SCRIPT_DIR = Path(__file__).resolve().parent
OUTPUT_DIR = SCRIPT_DIR / "downloaded_images"
TIMEOUT_SECONDS = 20
CHUNK_SIZE = 64 * 1024
NAME_LOCK = threading.Lock()

# Estensioni che possono essere ricavate in modo affidabile dal nome nell'URL.
IMAGE_EXTENSIONS = {
    ".avif",
    ".bmp",
    ".gif",
    ".heic",
    ".heif",
    ".ico",
    ".jfif",
    ".jpeg",
    ".jpg",
    ".png",
    ".svg",
    ".tif",
    ".tiff",
    ".webp",
}

CONTENT_TYPE_EXTENSIONS = {
    "image/avif": ".avif",
    "image/bmp": ".bmp",
    "image/gif": ".gif",
    "image/heic": ".heic",
    "image/heif": ".heif",
    "image/jpeg": ".jpg",
    "image/png": ".png",
    "image/svg+xml": ".svg",
    "image/tiff": ".tiff",
    "image/vnd.microsoft.icon": ".ico",
    "image/webp": ".webp",
    "image/x-icon": ".ico",
}


@dataclass(frozen=True)
class DownloadFailure:
    url: str
    reason: str


@dataclass(frozen=True)
class ImageJob:
    url: str
    image_index: int
    folder_slug: str | None = None


def validate_mongodb_config() -> None:
    values = {
        "MONGODB_URI": MONGODB_URI,
        "MONGODB_DATABASE": MONGODB_DATABASE,
        "MONGODB_COLLECTION": MONGODB_COLLECTION,
        "IMAGES_FIELD": IMAGES_FIELD,
        "DOCUMENT_SLUG_FIELD": DOCUMENT_SLUG_FIELD,
        "DOCUMENT_NAME_FIELD": DOCUMENT_NAME_FIELD,
    }
    invalid = [name for name, value in values.items() if not isinstance(value, str) or not value.strip()]
    if invalid:
        raise ValueError(f"configurazione MongoDB non valida: {', '.join(invalid)}")
    if not isinstance(MONGODB_TIMEOUT_MS, int) or MONGODB_TIMEOUT_MS <= 0:
        raise ValueError("MONGODB_TIMEOUT_MS deve essere un intero positivo")
    numeric_values = {
        "MONGODB_BATCH_SIZE": MONGODB_BATCH_SIZE,
        "DOWNLOAD_WORKERS": DOWNLOAD_WORKERS,
        "MAX_PENDING_DOWNLOADS": MAX_PENDING_DOWNLOADS,
    }
    invalid_numeric = [
        name
        for name, value in numeric_values.items()
        if not isinstance(value, int) or value <= 0
    ]
    if invalid_numeric:
        raise ValueError(
            "i seguenti parametri devono essere interi positivi: "
            + ", ".join(invalid_numeric)
        )
    if MAX_PENDING_DOWNLOADS < DOWNLOAD_WORKERS:
        raise ValueError(
            "MAX_PENDING_DOWNLOADS deve essere maggiore o uguale a DOWNLOAD_WORKERS"
        )


def document_folder_slug(document: dict) -> str:
    raw_slug = document.get(DOCUMENT_SLUG_FIELD)
    document_id = document.get("_id", "sconosciuto")
    if raw_slug is not None and not isinstance(raw_slug, str):
        raise ValueError(
            f"il campo '{DOCUMENT_SLUG_FIELD}' del documento {document_id} "
            "deve essere una stringa"
        )
    if isinstance(raw_slug, str) and raw_slug.strip():
        source = raw_slug
    else:
        name = document.get(DOCUMENT_NAME_FIELD)
        if not isinstance(name, str) or not name.strip():
            raise ValueError(
                f"il documento {document_id} non contiene né "
                f"'{DOCUMENT_SLUG_FIELD}' né un '{DOCUMENT_NAME_FIELD}' valido"
            )
        source = name
    try:
        return slugify(source)
    except ValueError as exc:
        raise ValueError(
            f"impossibile generare la cartella per il documento "
            f"{document_id}: {exc}"
        ) from exc


def iter_image_jobs_from_mongodb(client_factory=None) -> Iterator[ImageJob]:
    """Legge da MongoDB i download associati alle cartelle di destinazione."""
    validate_mongodb_config()
    if client_factory is None:
        try:
            from pymongo import MongoClient
            from pymongo.errors import PyMongoError
        except ImportError as exc:
            raise ValueError(
                "dipendenza mancante: installare pymongo con "
                "'python3 -m pip install -r requirements.txt'"
            ) from exc
        client_factory = MongoClient
    else:
        # Consente test senza richiedere una connessione o pymongo installato.
        PyMongoError = Exception

    client = None
    try:
        client = client_factory(MONGODB_URI, serverSelectionTimeoutMS=MONGODB_TIMEOUT_MS)
        client.admin.command("ping")
        collection = client[MONGODB_DATABASE][MONGODB_COLLECTION]
        documents = collection.find(
            {IMAGES_FIELD: {"$exists": True}},
            {
                IMAGES_FIELD: 1,
                DOCUMENT_SLUG_FIELD: 1,
                DOCUMENT_NAME_FIELD: 1,
            },
            batch_size=MONGODB_BATCH_SIZE,
        )

        for document in documents:
            folder_slug = document_folder_slug(document)
            gallery = document.get(IMAGES_FIELD)
            document_id = document.get("_id", "sconosciuto")
            if not isinstance(gallery, list):
                raise ValueError(
                    f"il campo '{IMAGES_FIELD}' del documento {document_id} "
                    "deve essere un array"
                )
            if any(not isinstance(url, str) or not url.strip() for url in gallery):
                raise ValueError(
                    f"il campo '{IMAGES_FIELD}' del documento {document_id} deve "
                    "contenere esclusivamente stringhe URL non vuote"
                )
            for image_index, url in enumerate(gallery, start=1):
                yield ImageJob(url.strip(), image_index, folder_slug)
    except ValueError:
        raise
    except PyMongoError as exc:
        raise ValueError(f"errore MongoDB: {exc}") from exc
    except Exception as exc:
        raise ValueError(f"impossibile leggere i dati da MongoDB: {exc}") from exc
    finally:
        if client is not None:
            client.close()


def load_urls_from_mongodb(client_factory=None) -> list[str]:
    """Compatibilità/test: materializza in una lista il flusso MongoDB."""
    return [job.url for job in iter_image_jobs_from_mongodb(client_factory)]


def slugify(value: str) -> str:
    """Converte il testo in uno slug ASCII adatto a un nome file."""
    normalized = unicodedata.normalize("NFKD", value)
    ascii_value = normalized.encode("ascii", "ignore").decode("ascii").lower()
    slug = re.sub(r"[^a-z0-9]+", "-", ascii_value).strip("-")
    if not slug:
        raise ValueError("lo slug non contiene lettere o numeri utilizzabili")
    return slug


def validate_url(url: str) -> None:
    parsed = urlparse(url)
    if parsed.scheme not in {"http", "https"} or not parsed.netloc:
        raise ValueError("URL non valido: sono supportati soltanto URL HTTP/HTTPS")


def content_type_extension(content_type: str) -> str:
    media_type = content_type.partition(";")[0].strip().lower()
    extension = CONTENT_TYPE_EXTENSIONS.get(media_type)
    if extension:
        return extension
    guessed = mimetypes.guess_extension(media_type, strict=False)
    return guessed or ".img"


def url_filename_parts(url: str, index: int, content_type: str) -> tuple[str, str]:
    """Restituisce nome base sicuro ed estensione per un URL senza slug."""
    url_name = Path(unquote(urlparse(url).path)).name
    url_path = Path(url_name)
    url_extension = url_path.suffix.lower()
    extension = (
        url_extension
        if url_extension in IMAGE_EXTENSIONS
        else content_type_extension(content_type)
    )

    raw_stem = url_path.stem if url_extension in IMAGE_EXTENSIONS else url_name
    # Rimuove caratteri vietati su Windows e caratteri di controllo.
    stem = re.sub(r'[<>:"/\\|?*\x00-\x1f]', "_", raw_stem).strip(" .")
    if not stem:
        stem = f"immagine-{index}"
    return stem, extension


def available_path(directory: Path, stem: str, extension: str) -> Path:
    """Trova un nome libero aggiungendo _1, _2, ... prima dell'estensione."""
    candidate = directory / f"{stem}{extension}"
    suffix = 1
    while candidate.exists():
        candidate = directory / f"{stem}_{suffix}{extension}"
        suffix += 1
    return candidate


def reserve_path(directory: Path, stem: str, extension: str) -> Path:
    """Prenota un nome libero impedendo collisioni tra download concorrenti."""
    with NAME_LOCK:
        while True:
            candidate = available_path(directory, stem, extension)
            try:
                candidate.touch(exist_ok=False)
            except FileExistsError:
                continue
            return candidate


def download_one(
    url: str,
    index: int,
    output_dir: Path,
    common_slug: str | None,
    timeout: float = TIMEOUT_SECONDS,
) -> Path:
    validate_url(url)
    request = Request(url, headers={"User-Agent": "download-images/1.0"})

    with urlopen(request, timeout=timeout) as response:
        content_type = response.headers.get("Content-Type", "")
        media_type = content_type.partition(";")[0].strip().lower()
        if not media_type.startswith("image/"):
            shown_type = media_type or "assente"
            raise ValueError(f"contenuto non immagine (Content-Type: {shown_type})")

        extension = content_type_extension(content_type)
        if common_slug is not None:
            stem = f"{common_slug}-{index}"
        else:
            stem, extension = url_filename_parts(url, index, content_type)

        destination = reserve_path(output_dir, stem, extension)
        temporary = destination.with_name(f".{destination.name}.part")
        try:
            with temporary.open("wb") as file:
                shutil.copyfileobj(response, file, length=CHUNK_SIZE)
            temporary.replace(destination)
        except Exception:
            destination.unlink(missing_ok=True)
            raise
        finally:
            temporary.unlink(missing_ok=True)

    return destination


def _download_failure(url: str, exc: Exception) -> DownloadFailure:
    if isinstance(exc, HTTPError):
        reason = f"errore HTTP {exc.code}"
    elif isinstance(exc, URLError):
        reason = f"errore di rete: {exc.reason}"
    elif isinstance(exc, (OSError, TimeoutError, ValueError)):
        reason = str(exc)
    else:
        reason = f"errore inatteso: {exc}"
    return DownloadFailure(url, reason)


def download_all(
    urls: Iterable[str | ImageJob],
    output_dir: Path,
    common_slug: str | None,
    workers: int = DOWNLOAD_WORKERS,
    max_pending: int = MAX_PENDING_DOWNLOADS,
) -> tuple[list[Path], list[DownloadFailure]]:
    if workers <= 0 or max_pending < workers:
        raise ValueError("configurazione del pool di download non valida")
    output_dir.mkdir(parents=True, exist_ok=True)
    downloaded_indexed: list[tuple[int, Path]] = []
    failed_indexed: list[tuple[int, DownloadFailure]] = []
    numbered_urls = enumerate(urls, start=1)
    pending: dict[Future[Path], tuple[int, str]] = {}
    exhausted = False

    with ThreadPoolExecutor(max_workers=workers) as executor:
        while pending or not exhausted:
            while not exhausted and len(pending) < max_pending:
                try:
                    order, item = next(numbered_urls)
                except StopIteration:
                    exhausted = True
                    break
                if isinstance(item, ImageJob):
                    url = item.url
                    image_index = item.image_index
                    destination_dir = (
                        output_dir / item.folder_slug
                        if item.folder_slug is not None
                        else output_dir
                    )
                else:
                    url = item
                    image_index = order
                    destination_dir = output_dir
                destination_dir.mkdir(parents=True, exist_ok=True)
                future = executor.submit(
                    download_one,
                    url,
                    image_index,
                    destination_dir,
                    common_slug,
                )
                pending[future] = (order, url)

            if not pending:
                continue

            completed, _ = wait(pending, return_when=FIRST_COMPLETED)
            for future in completed:
                index, url = pending.pop(future)
                try:
                    destination = future.result()
                except Exception as exc:
                    failed_indexed.append((index, _download_failure(url, exc)))
                else:
                    downloaded_indexed.append((index, destination))

    downloaded_indexed.sort(key=lambda item: item[0])
    failed_indexed.sort(key=lambda item: item[0])
    return (
        [path for _, path in downloaded_indexed],
        [failure for _, failure in failed_indexed],
    )


def print_summary(downloaded: Sequence[Path], failed: Sequence[DownloadFailure]) -> None:
    print("\nRiepilogo")
    print(f"  File scaricati: {len(downloaded)}")
    print(f"  URL falliti: {len(failed)}")
    for failure in failed:
        print(f"    - {failure.url} ({failure.reason})")


def parse_args(argv: Sequence[str] | None = None) -> argparse.Namespace:
    parser = argparse.ArgumentParser(
        description="Scarica le immagini elencate nei documenti MongoDB."
    )
    parser.add_argument(
        "slug",
        nargs="?",
        help="nome comune opzionale usato per numerare le immagini",
    )
    return parser.parse_args(argv)


def main(argv: Sequence[str] | None = None) -> int:
    args = parse_args(argv)
    try:
        common_slug = slugify(args.slug) if args.slug is not None else None
        jobs = iter_image_jobs_from_mongodb()
        downloaded, failed = download_all(jobs, OUTPUT_DIR, common_slug)
    except ValueError as exc:
        print(f"Errore: {exc}", file=sys.stderr)
        return 2

    print_summary(downloaded, failed)
    return 1 if failed else 0


if __name__ == "__main__":
    raise SystemExit(main())
