#!/usr/bin/env python3
"""
Transcription automatique de vidéos MP4 avec faster-whisper.
- Transcrit en français
- Génère un fichier .txt avec timestamps dans un sous-dossier 'transcripts/'
- Génère aussi un fichier .html compatible NGINX (encodage UTF-8 propre, pas de bugs accents)
- Ignore les silences et répétitions (via VAD)
- Tourne sur CPU uniquement
- Ne retranscrit pas une vidéo déjà traitée (vérifie l'existence du fichier transcript)
- S'arrête après avoir traité toutes les vidéos, ne tourne pas en boucle

Usage :
    python transcribe.py [dossier_videos]

    Si aucun dossier n'est fourni, utilise le dossier courant.

Installation des dépendances :
    pip install faster-whisper
"""

import os
import sys
import time
from pathlib import Path


def format_timestamp(seconds: float) -> str:
    """Convertit des secondes en format HH:MM:SS,mmm"""
    h = int(seconds // 3600)
    m = int((seconds % 3600) // 60)
    s = int(seconds % 60)
    ms = int((seconds % 1) * 1000)
    return f"{h:02d}:{m:02d}:{s:02d},{ms:03d}"


def is_repetition_or_filler(text: str) -> bool:
    """
    Détecte les répétitions, hésitations et bruits parasites à ignorer.
    faster-whisper avec VAD filtre déjà la plupart, mais on ajoute une couche.
    """
    text = text.strip().lower()

    # Ignorer les segments trop courts ou vides
    if len(text) <= 1:
        return True

    # Hésitations et bruits typiques
    fillers = {
        "euh", "euhm", "hm", "hmm", "hmmm", "ah", "ahh", "oh",
        "um", "umm", "uh", "uhh", "mhm", "mmh", "mmm",
        "[musique]", "[music]", "[bruit]", "[silence]", "[applaudissements]",
        "(musique)", "(music)", "(bruit)", "(silence)",
        "...", "…",
    }

    if text in fillers:
        return True

    # Ignorer si le texte ne contient que des caractères non-alphabétiques
    alpha_chars = [c for c in text if c.isalpha()]
    if len(alpha_chars) < 2:
        return True

    return False


def transcribe_video(video_path: Path, transcripts_dir: Path, model) -> bool:
    """
    Transcrit une vidéo et génère les fichiers de sortie.
    Retourne True si succès, False si erreur.
    """
    stem = video_path.stem
    txt_path = transcripts_dir / f"{stem}.txt"
    html_path = transcripts_dir / f"{stem}.html"

    # Vérifier si le transcript existe déjà
    if txt_path.exists():
        print(f"  [SKIP] Transcript déjà existant : {txt_path.name}")
        return True

    print(f"  [TRAITEMENT] {video_path.name}")
    print(f"    Transcription en cours (CPU, cela peut prendre du temps pour les longues vidéos)...")

    start_time = time.time()

    try:
        # Transcription avec VAD activé pour filtrer les silences
        segments, info = model.transcribe(
            str(video_path),
            language="fr",
            beam_size=5,
            vad_filter=True,                  # Filtre Voice Activity Detection
            vad_parameters={
                "min_silence_duration_ms": 500,   # Ignore les silences > 500ms
                "speech_pad_ms": 200,             # Padding autour de la parole
                "threshold": 0.5,                 # Seuil de détection de parole
            },
            word_timestamps=False,
            condition_on_previous_text=True,      # Cohérence contextuelle
            compression_ratio_threshold=2.4,      # Filtre les segments répétitifs
            no_speech_threshold=0.6,              # Seuil silence/parole
            log_prob_threshold=-1.0,
        )

        # Collecter les segments filtrés
        valid_segments = []
        prev_text = ""

        for seg in segments:
            text = seg.text.strip()

            # Filtrer les répétitions et bruits
            if is_repetition_or_filler(text):
                continue

            # Filtrer les répétitions exactes consécutives
            if text.lower() == prev_text.lower():
                continue

            valid_segments.append({
                "start": seg.start,
                "end": seg.end,
                "text": text,
            })
            prev_text = text

        elapsed = time.time() - start_time
        print(f"    Terminé en {elapsed:.1f}s — {len(valid_segments)} segments")

        # --- Écriture du fichier TXT avec timestamps ---
        with open(txt_path, "w", encoding="utf-8") as f:
            f.write(f"Transcription : {video_path.name}\n")
            f.write(f"Langue détectée : {info.language} (probabilité : {info.language_probability:.2%})\n")
            f.write(f"Durée : {format_timestamp(info.duration)}\n")
            f.write("=" * 60 + "\n\n")

            for seg in valid_segments:
                ts_start = format_timestamp(seg["start"])
                ts_end = format_timestamp(seg["end"])
                f.write(f"[{ts_start} --> {ts_end}]\n")
                f.write(f"{seg['text']}\n\n")

        print(f"    TXT sauvegardé : {txt_path}")

        # --- Écriture du fichier HTML compatible NGINX (UTF-8, accents OK) ---
        with open(html_path, "w", encoding="utf-8") as f:
            f.write('<!DOCTYPE html>\n')
            f.write('<html lang="fr">\n')
            f.write('<head>\n')
            f.write('  <meta charset="UTF-8">\n')
            f.write('  <meta name="viewport" content="width=device-width, initial-scale=1.0">\n')
            f.write(f'  <title>Transcription — {html_escape(video_path.name)}</title>\n')
            f.write('  <style>\n')
            f.write('    body { font-family: sans-serif; max-width: 900px; margin: 2rem auto; padding: 0 1rem; line-height: 1.6; }\n')
            f.write('    h1 { font-size: 1.4rem; border-bottom: 2px solid #333; padding-bottom: .5rem; }\n')
            f.write('    .meta { color: #555; font-size: .9rem; margin-bottom: 1.5rem; }\n')
            f.write('    .segment { margin-bottom: 1rem; }\n')
            f.write('    .timestamp { font-family: monospace; font-size: .8rem; color: #888; display: block; }\n')
            f.write('    .text { margin: .2rem 0 0 0; }\n')
            f.write('  </style>\n')
            f.write('</head>\n')
            f.write('<body>\n')
            f.write(f'  <h1>Transcription : {html_escape(video_path.name)}</h1>\n')
            f.write(f'  <p class="meta">'
                    f'Langue : {html_escape(info.language)} ({info.language_probability:.2%}) &nbsp;|&nbsp; '
                    f'Durée : {html_escape(format_timestamp(info.duration))}'
                    f'</p>\n')

            for seg in valid_segments:
                ts_start = format_timestamp(seg["start"])
                ts_end = format_timestamp(seg["end"])
                f.write('  <div class="segment">\n')
                f.write(f'    <span class="timestamp">[{html_escape(ts_start)} &#8594; {html_escape(ts_end)}]</span>\n')
                f.write(f'    <p class="text">{html_escape(seg["text"])}</p>\n')
                f.write('  </div>\n')

            f.write('</body>\n')
            f.write('</html>\n')

        print(f"    HTML sauvegardé : {html_path}")
        return True

    except Exception as e:
        print(f"    [ERREUR] {e}")
        # Supprimer les fichiers partiels en cas d'erreur
        for p in [txt_path, html_path]:
            if p.exists():
                p.unlink()
        return False


def html_escape(text: str) -> str:
    """Échappe les caractères spéciaux HTML (accents gérés via UTF-8, seuls <>&\" sont échappés)."""
    return (
        text
        .replace("&", "&amp;")
        .replace("<", "&lt;")
        .replace(">", "&gt;")
        .replace('"', "&quot;")
    )


def main():
    # Dossier source : argument CLI ou dossier courant
    if len(sys.argv) > 1:
        video_dir = Path(sys.argv[1])
    else:
        video_dir = Path(".")

    if not video_dir.is_dir():
        print(f"[ERREUR] Le dossier '{video_dir}' n'existe pas.")
        sys.exit(1)

    # Trouver toutes les vidéos MP4
    videos = sorted(video_dir.glob("*.mp4"))
    if not videos:
        print(f"Aucun fichier .mp4 trouvé dans '{video_dir}'.")
        sys.exit(0)

    print(f"Dossier vidéos   : {video_dir.resolve()}")
    print(f"Vidéos trouvées  : {len(videos)}")

    # Créer le dossier transcripts à côté du script (là où il se trouve)
    script_dir = Path(__file__).parent.resolve()
    transcripts_dir = script_dir / "transcripts"
    transcripts_dir.mkdir(exist_ok=True)
    print(f"Dossier transcripts : {transcripts_dir}")
    print()

    # Chargement du modèle (une seule fois)
    print("Chargement du modèle faster-whisper (large-v3, CPU)...")
    print("(Le premier chargement télécharge le modèle si nécessaire — ~3 Go)\n")

    try:
        from faster_whisper import WhisperModel
    except ImportError:
        print("[ERREUR] faster-whisper n'est pas installé.")
        print("Installez-le avec : pip install faster-whisper")
        sys.exit(1)

    # CPU, int8 pour réduire la RAM et accélérer sur CPU
    model = WhisperModel(
        "large-v3",
        device="cpu",
        compute_type="int8",   # int8 : 2x plus rapide sur CPU, qualité quasi identique
        cpu_threads=os.cpu_count() or 4,
        num_workers=1,
    )

    print("Modèle chargé.\n")
    print("=" * 60)

    # Traitement de chaque vidéo
    ok = 0
    skipped = 0
    errors = 0

    for i, video in enumerate(videos, 1):
        print(f"[{i}/{len(videos)}] {video.name}")
        stem = video.stem
        txt_path = transcripts_dir / f"{stem}.txt"

        if txt_path.exists():
            print(f"  [SKIP] Transcript existant, vidéo ignorée.")
            skipped += 1
        else:
            success = transcribe_video(video, transcripts_dir, model)
            if success:
                ok += 1
            else:
                errors += 1
        print()

    print("=" * 60)
    print(f"Terminé.")
    print(f"  Transcrites : {ok}")
    print(f"  Ignorées (déjà faites) : {skipped}")
    print(f"  Erreurs : {errors}")
    print(f"  Fichiers dans : {transcripts_dir}")


if __name__ == "__main__":
    main()
