"""
Extraction du texte des PDF pour alimenter l'IA (chatbot Goonie, génération de quiz).
"""
import logging
from pathlib import Path

from django.conf import settings

logger = logging.getLogger(__name__)


def extract_text_from_pdf(file_path_or_file):
    """
    Extrait le texte d'un fichier PDF.
    file_path_or_file : chemin (str/Path) ou objet File Django (avec .path, .name ou .open()).
    Retourne une chaîne (vide si échec).
    """
    try:
        if hasattr(file_path_or_file, "open") and callable(getattr(file_path_or_file, "open", None)):
            # FileField Django : privilégier .path si dispo, sinon .open("rb")
            if hasattr(file_path_or_file, "path"):
                try:
                    path = Path(file_path_or_file.path)
                    if path.exists():
                        with open(path, "rb") as f:
                            return _extract_text_from_stream(f)
                except Exception:
                    pass
            # Fallback : lire via .open() (marche aussi si stockage distant)
            with file_path_or_file.open("rb") as f:
                return _extract_text_from_stream(f)

        path = Path(file_path_or_file)
        if path.exists():
            with open(path, "rb") as f:
                return _extract_text_from_stream(f)
        # Essayer MEDIA_ROOT + nom relatif (ex. "lessons/pdfs/xxx.pdf")
        if hasattr(file_path_or_file, "name"):
            alt = Path(settings.MEDIA_ROOT) / file_path_or_file.name
            if alt.exists():
                with open(alt, "rb") as f:
                    return _extract_text_from_stream(f)
        logger.warning("[PDF] Fichier introuvable: %s", path)
        return ""
    except Exception as e:
        logger.exception("[PDF] Erreur extraction: %s", e)
        return ""


def _extract_text_from_stream(file_handle):
    from pypdf import PdfReader

    reader = PdfReader(file_handle)
    parts = []
    for i, page in enumerate(reader.pages):
        try:
            text = page.extract_text()
            if text:
                parts.append(text.strip())
        except Exception as e:
            logger.warning("[PDF] Page %s: %s", i + 1, e)
    return "\n\n".join(parts) if parts else ""
