"""
services/video_engine.py — Custom Wav2Lip-based lip-sync video engine.

Replaces the prior Tavus integration. The role-play WebSocket calls
`generate_lipsync_video()` once per AI turn (after TTS audio is fully
streamed) to produce an MP4 of one of our local avatars lip-syncing to
the synthesized speech.

──────────────────────────────────────────────────────────────────────
Why subprocess + cache + lock
──────────────────────────────────────────────────────────────────────
* Wav2Lip's `inference.py` is a CLI; calling it as a subprocess avoids
  importing torch / cv2 / librosa into our FastAPI process — those deps
  are heavy, slow to import, and would brittle the entire app if the
  install were broken. The route stays loadable even if Wav2Lip is
  missing; it only fails at call time, with a clear error.
* On CPU, a single Wav2Lip render takes 30–90 s. Caching by the SHA-256
  of (audio_bytes + avatar) means a candidate hearing the same scripted
  greeting twice waits only on the first turn — every replay is instant.
* An asyncio.Lock serialises inference so two concurrent sessions don't
  thrash the CPU. Conversations stay responsive (the second user just
  waits their turn rather than getting a half-speed render).
──────────────────────────────────────────────────────────────────────
Setup
──────────────────────────────────────────────────────────────────────
Run `bash scripts/setup_wav2lip.sh` once on the server. It:
  • clones Wav2Lip into  vendor/Wav2Lip/
  • installs the Python deps into the active venv
  • downloads the pretrained model into  vendor/Wav2Lip/checkpoints/
  • verifies ffmpeg is on PATH

Place your 4 avatar MP4s in:
  static/assets/avatars/male_one.mp4
  static/assets/avatars/male_two.mp4
  static/assets/avatars/female_one.mp4
  static/assets/avatars/female_two.mp4

Generated videos go in:
  data/cache/video/<sha256>.mp4
and are served at  /static/cache/video/<sha256>.mp4  via the static
mount in main.py.
"""
from __future__ import annotations

import asyncio
import hashlib
import logging
import os
import random
import shutil
import struct
import subprocess
import wave
from pathlib import Path
from typing import Optional

log = logging.getLogger(__name__)

# ── Paths ────────────────────────────────────────────────────────────
BASE_DIR        = Path(__file__).resolve().parent.parent
WAV2LIP_DIR     = BASE_DIR / "vendor" / "Wav2Lip"
WAV2LIP_CKPT    = WAV2LIP_DIR / "checkpoints" / "wav2lip_gan.pth"
AVATAR_DIR      = BASE_DIR / "static" / "assets" / "avatars"
CACHE_DIR       = BASE_DIR / "data" / "cache" / "video"
TMP_DIR         = BASE_DIR / "data" / "cache" / "tmp"

# Ensure cache + tmp dirs exist on import (cheap, idempotent).
for _d in (CACHE_DIR, TMP_DIR):
    _d.mkdir(parents=True, exist_ok=True)

# ── Avatar registry ─────────────────────────────────────────────────
# All four avatars are equally likely within their gender bucket.
# Adding/removing files in the AVATAR_DIR is picked up automatically —
# no code change needed. To add new avatars: drop the MP4 into
# static/assets/avatars/ AND add an entry to AVATAR_GENDER below so
# the voice-matcher knows which OpenAI TTS voice to pair it with.
AVATAR_FILENAMES: list[str] = [
    "female_one.mp4",
    "female_two.mp4",

    "female_three.mp4",

    "female_four.mp4",

    "female_five.mp4",

    "female_six.mp4",

    "female_seven.mp4",

    "female_eight.mp4",

    "female_nine.mp4",

    "female_ten.mp4",

    "male_one.mp4",

    "male_two.mp4",

    "male_three.mp4",

    "male_four.mp4",

    "male_five.mp4",

    "male_six.mp4",

    "male_seven.mp4",

    "male_eight.mp4",

    "male_nine.mp4",

    "male_ten.mp4",

]

# Filename → gender. Drives the voice-pairing logic so the candidate
# never hears a female voice over a male avatar (or vice versa).
# Files not in this map default to "male" — that's a deliberate
# conservative fallback so we never accidentally lip-sync a feminine
# voice onto an unknown face. The right way to add a new avatar is
# to put its filename here at the same time as dropping it on disk.
AVATAR_GENDER: dict[str, str] = {
    "female_one.mp4":   "female",

    "female_two.mp4":   "female",

    "female_three.mp4": "female",

    "female_four.mp4":  "female",

    "female_five.mp4":  "female",

    "female_six.mp4":   "female",

    "female_seven.mp4": "female",

    "female_eight.mp4": "female",

    "female_nine.mp4":  "female",

    "female_ten.mp4":   "female",

    "male_one.mp4":   "male",

    "male_two.mp4":   "male",

    "male_three.mp4": "male",

    "male_four.mp4":  "male",

    "male_five.mp4":  "male",

    "male_six.mp4":   "male",

    "male_seven.mp4": "male",

    "male_eight.mp4": "male",

    "male_nine.mp4":  "male",

    "male_ten.mp4":   "male",

}

# Gender → OpenAI TTS voice. These are the four voices that read most
# clearly as one gender or the other in user testing:
#   • onyx   — deep, mature male
#   • fable  — softer male (used as the second male voice if you want
#              variety between sessions; not used today)
#   • nova   — warm, professional female
#   • shimmer — brighter female (alternate; not used today)
# Mapping is deliberately 1:1 today so the same gender always uses
# the same voice — predictability matters more than variety. Change
# the right-hand side if your scenario demands a different timbre.
VOICE_BY_GENDER: dict[str, str] = {
    "male":   "onyx",
    "female": "nova",
}


def list_available_avatars() -> list[str]:
    """Return the names of avatar files actually present on disk.

    Discovery is now DYNAMIC — we scan static/assets/avatars/ for any
    .mp4 file rather than relying on the hardcoded AVATAR_FILENAMES.
    That means an operator can drop new avatar clips into the folder
    and they show up in the picker / random pool without a code
    change (per the "scalable so new avatar videos automatically
    appear" requirement on the avatar-selection feature).

    The hardcoded list is still used as a fall-through if the
    directory is missing or empty, so dev environments without the
    asset bundle still boot.
    """
    if AVATAR_DIR.is_dir():
        found = sorted(p.name for p in AVATAR_DIR.glob("*.mp4") if p.is_file())
        if found:
            return found
    return [
        name for name in AVATAR_FILENAMES
        if (AVATAR_DIR / name).is_file()
    ]


def _gender_from_filename(name: str) -> str:
    """Infer a gender label from a filename. Conventional names are
    `male_*.mp4` / `female_*.mp4`. Anything else falls back to "male"
    as a conservative default (same fallback as AVATAR_GENDER for
    unknown keys)."""
    if not name:
        return "male"
    lower = name.lower()
    # Explicit map wins so curated overrides (e.g. a non-standard
    # filename pinned to a specific gender) keep working.
    if name in AVATAR_GENDER:
        return AVATAR_GENDER[name]
    if lower.startswith("female"):
        return "female"
    if lower.startswith("male"):
        return "male"
    return "male"


def avatar_manifest() -> list[dict]:
    """Return one dict per available avatar:
        { name, gender, size_bytes, url }
    Used by the GET /api/avatars endpoint that powers the candidate-
    facing avatar picker modal. Names are sorted so the picker order
    is stable between calls.
    """
    items: list[dict] = []
    for name in list_available_avatars():
        path = AVATAR_DIR / name
        try:
            size = path.stat().st_size if path.is_file() else 0
        except Exception:  # noqa: BLE001
            size = 0
        items.append({
            "name":       name,
            "gender":     _gender_from_filename(name),
            "size_bytes": size,
            "url":        f"/static/assets/avatars/{name}",
            "voice":      voice_for_avatar(name),
        })
    return items


def avatar_gender(avatar_name: str) -> str:
    """Return 'male' or 'female' for a given avatar filename.

    Falls back to 'male' for unknown filenames — see the comment on
    AVATAR_GENDER. The fallback is conservative on purpose: a wrong
    gender produces the bug we're trying to fix, so we'd rather have
    an unknown avatar default to a single consistent voice than have
    it picked at random.
    """
    return AVATAR_GENDER.get(avatar_name, "male")


def voice_for_avatar(avatar_name: str) -> str:
    """Return the OpenAI TTS voice ID that matches this avatar's
    gender. The role-play WebSocket calls this immediately after
    picking the avatar and threads the result through every
    stream_tts_chunks() call for the session, so audio and video
    always agree on gender."""
    return VOICE_BY_GENDER.get(avatar_gender(avatar_name), VOICE_BY_GENDER["male"])


def pick_avatar(seed: Optional[str] = None, gender: Optional[str] = None) -> str:
    """Return a random avatar filename (e.g. 'male_one.mp4').

    Pass ``seed`` (e.g. the session_id) to get a deterministic pick for
    a given session — the same session always uses the same avatar so
    the candidate doesn't see the face change mid-conversation. Calling
    again with the same seed returns the same avatar.

    Pass ``gender`` ('male' or 'female') to restrict the pool to
    avatars of that gender. When omitted the full pool is used. This
    is the seam used by scenarios that want to lock the interviewer's
    gender (e.g. ``ai_gender`` on the scenario JSON).
    """
    available = list_available_avatars() or AVATAR_FILENAMES
    if gender:
        gender = gender.strip().lower()
        pool = [a for a in available if avatar_gender(a) == gender]
        # Empty filtered pool → fall back to the full list so a typo on
        # the scenario doesn't 500 the session.
        if not pool:
            pool = available
    else:
        pool = available
    if seed:
        rng = random.Random(seed)
        return rng.choice(pool)
    return random.choice(pool)


# ── Concurrency control ─────────────────────────────────────────────
# Wav2Lip is single-threaded on CPU; running two at once just halves
# each render's speed. Serialise instead.
_inference_lock = asyncio.Lock()


# ── GPU detection ───────────────────────────────────────────────────
# On Wav2Lip the difference between CPU and CUDA is roughly 10× —
# 30-90 s/turn vs 3-8 s/turn for a typical 4-second utterance. We
# detect by:
#   1. Honouring an explicit override: WAV2LIP_USE_GPU=true / false.
#   2. Otherwise calling nvidia-smi once at import (cached). If it
#      exits 0 we assume CUDA is wired through to whatever Python
#      Wav2Lip will use.
# The result is cached so we don't shell out for every turn.
_GPU_CACHED: Optional[bool] = None


# ── Warmup ──────────────────────────────────────────────────────────
# Wav2Lip's inference.py imports torch / cv2 / face_detection / librosa
# on every invocation. On cold start those imports cost 5-12 s — fully
# half the perceived "first AI turn" latency on a CPU box. We do a
# zero-work import-only warmup once at app boot so subsequent calls
# only pay the actual inference cost.
_WARMUP_STARTED = False


def warmup_wav2lip_async() -> None:
    """Fire-and-forget warmup. Safe to call from a sync context (e.g.
    FastAPI lifespan); we kick off the subprocess and don't wait for
    it. Idempotent — only the first call does anything."""
    global _WARMUP_STARTED
    if _WARMUP_STARTED:
        return
    _WARMUP_STARTED = True
    if not is_wav2lip_installed():
        log.info("[video_engine] warmup skipped — Wav2Lip not installed")
        return
    try:
        # Tiny inline script: import the heavy modules so the file
        # cache and Python bytecode are primed, then exit. We DO NOT
        # run actual inference (that would burn cycles without an
        # input audio file). The next real render still does its
        # own import, but those imports hit a hot file cache and
        # come back in milliseconds instead of seconds.
        script = (
            "import sys, os\n"
            f"sys.path.insert(0, r'{WAV2LIP_DIR}')\n"
            "try:\n"
            "  import torch, cv2, librosa, numpy\n"
            "  import face_detection  # type: ignore\n"
            "  print('[video_engine] warmup imports OK')\n"
            "except Exception as e:\n"
            "  print('[video_engine] warmup import error:', e)\n"
        )
        subprocess.Popen(
            ["python", "-c", script],
            cwd=str(WAV2LIP_DIR),
            stdout=subprocess.DEVNULL,
            stderr=subprocess.DEVNULL,
        )
        log.info("[video_engine] Wav2Lip warmup subprocess launched")
    except Exception as exc:  # noqa: BLE001
        log.warning("[video_engine] warmup failed (ignored): %s", exc)


def _should_use_wav2lip_gpu() -> bool:
    global _GPU_CACHED
    if _GPU_CACHED is not None:
        return _GPU_CACHED
    # Explicit override wins.
    env = (os.getenv("WAV2LIP_USE_GPU", "") or "").strip().lower()
    if env in ("1", "true", "yes", "on"):
        _GPU_CACHED = True
        return True
    if env in ("0", "false", "no", "off"):
        _GPU_CACHED = False
        return False
    # Auto-detect via nvidia-smi. If anything goes wrong fall through
    # to CPU — Wav2Lip's inference.py handles either way.
    if shutil.which("nvidia-smi") is None:
        _GPU_CACHED = False
        return False
    try:
        result = subprocess.run(
            ["nvidia-smi", "-L"],
            stdout=subprocess.PIPE,
            stderr=subprocess.PIPE,
            timeout=2.0,
            check=False,
        )
        _GPU_CACHED = (result.returncode == 0 and b"GPU" in (result.stdout or b""))
    except Exception:  # noqa: BLE001
        _GPU_CACHED = False
    if _GPU_CACHED:
        log.info("[video_engine] CUDA GPU detected — Wav2Lip will run on GPU.")
    else:
        log.info("[video_engine] No CUDA GPU detected — Wav2Lip will run on CPU.")
    return _GPU_CACHED


# ── Cache ────────────────────────────────────────────────────────────
def _cache_key(audio_bytes: bytes, avatar_name: str) -> str:
    """Hash the audio + avatar pair. Identical (audio, avatar) → same key
    → reuse the cached MP4. We hash the raw audio rather than the prompt
    text because TTS is not strictly deterministic (voice / prosody
    settings can drift) and we want the cache to reflect bit-identical
    output, not best-effort text matches."""
    h = hashlib.sha256()
    h.update(audio_bytes)
    h.update(b"::")
    h.update(avatar_name.encode("utf-8"))
    return h.hexdigest()


def _cached_video_path(key: str) -> Path:
    return CACHE_DIR / f"{key}.mp4"


def cached_url_for(key: str) -> str:
    """URL the frontend should load. Served by the /static/cache mount
    declared in main.py."""
    return f"/static/cache/video/{key}.mp4"


# ── Audio helpers ────────────────────────────────────────────────────
TTS_SAMPLE_RATE = 24_000  # matches services.openai_service.stream_tts_chunks
TTS_SAMPLE_WIDTH = 2      # 16-bit PCM


def trim_trailing_silence(pcm: bytes,
                          sample_rate: int = TTS_SAMPLE_RATE,
                          sample_width: int = TTS_SAMPLE_WIDTH,
                          threshold: int = 500,
                          min_keep_ms: int = 80) -> bytes:
    """Strip silent samples from the END of a PCM-16-LE buffer.

    Why this matters: TTS often appends 100–400 ms of low-amplitude
    tail. Wav2Lip animates a mouth shape for every audio frame, so
    that tail makes the avatar's lips keep moving for a beat after
    the voice has stopped — the exact "trailing animation" the user
    flagged. Trimming the silence before render makes the video end
    on the last spoken syllable.

    `threshold`   — amplitude below which a sample counts as silence
                    (16-bit PCM, so range is ±32 768; 500 ≈ -36 dBFS).
    `min_keep_ms` — never trim below this length even if everything
                    looks silent (defensive — keeps the engine sane
                    on extremely short utterances).
    """
    if not pcm:
        return pcm
    samples = len(pcm) // sample_width
    if samples == 0:
        return pcm
    # Walk backwards from the end one sample at a time looking for the
    # last sample whose magnitude exceeds the silence threshold.
    last_voiced = -1
    for i in range(samples - 1, -1, -1):
        off = i * sample_width
        # Little-endian signed 16-bit.
        val = int.from_bytes(pcm[off:off + sample_width], "little", signed=True)
        if abs(val) >= threshold:
            last_voiced = i
            break
    if last_voiced < 0:
        return pcm  # all silence — leave as-is; caller decides what to do
    # Keep ~40 ms of breathing room past the last voiced sample so the
    # final phoneme isn't clipped, but no more — we want the lips to
    # close, not keep mouthing.
    pad_samples = int(sample_rate * 0.04)
    keep_samples = min(samples, last_voiced + 1 + pad_samples)
    keep_samples = max(keep_samples, int(sample_rate * min_keep_ms / 1000))
    return pcm[: keep_samples * sample_width]


def pcm_to_wav_bytes(pcm: bytes,
                     sample_rate: int = TTS_SAMPLE_RATE,
                     sample_width: int = TTS_SAMPLE_WIDTH) -> bytes:
    """Wrap raw PCM-16-LE bytes in a WAV header. Wav2Lip's audio loader
    (librosa) needs a real audio container — it can't parse headerless
    PCM directly."""
    import io as _io
    buf = _io.BytesIO()
    with wave.open(buf, "wb") as wf:
        wf.setnchannels(1)
        wf.setsampwidth(sample_width)
        wf.setframerate(sample_rate)
        wf.writeframes(pcm)
    return buf.getvalue()


# ── Audio resampling for Wav2Lip ─────────────────────────────────────
# Wav2Lip's mel encoder was trained on 16 kHz mono. Feeding it 24 kHz
# audio (what OpenAI TTS produces) shifts the mel-bin distribution
# enough that the phoneme→viseme mapping subtly mis-times — the mouth
# articulates a hair AHEAD of the audio, which reads on screen as
# "the lips don't quite match". The fix is to resample down to 16 kHz
# BEFORE handing the file to inference.py. We do this with ffmpeg (no
# extra Python deps) so the path stays the same on every host.
WAV2LIP_NATIVE_SAMPLE_RATE = 16_000


async def _resample_wav_to_16k(src_path: Path, dst_path: Path) -> None:
    """Resample any WAV/PCM file to 16 kHz mono PCM-16-LE. No-op if the
    input is already 16 kHz mono; ffmpeg is smart enough to copy in
    that case. We always go through ffmpeg rather than librosa here
    because Wav2Lip's own dependency tree already pulls in ffmpeg
    and we don't want to add another resampler that might disagree.
    """
    if shutil.which("ffmpeg") is None:
        raise RuntimeError("ffmpeg is required for Wav2Lip audio resampling.")
    cmd = [
        "ffmpeg", "-y",
        "-i",  str(src_path),
        "-ac", "1",                              # mono
        "-ar", str(WAV2LIP_NATIVE_SAMPLE_RATE),  # 16 kHz
        "-c:a", "pcm_s16le",                     # match Wav2Lip's WAV reader
        "-loglevel", "error",
        str(dst_path),
    ]
    proc = await asyncio.create_subprocess_exec(
        *cmd,
        stdout=asyncio.subprocess.PIPE,
        stderr=asyncio.subprocess.PIPE,
    )
    _, stderr = await proc.communicate()
    if proc.returncode != 0:
        err_tail = (stderr or b"").decode("utf-8", "ignore")[-1500:]
        log.error("[video_engine] ffmpeg resample failed rc=%s stderr=\n%s",
                  proc.returncode, err_tail)
        raise RuntimeError(f"ffmpeg resample failed (exit {proc.returncode}).")
    if not dst_path.is_file() or dst_path.stat().st_size == 0:
        raise RuntimeError("ffmpeg resample produced no output file.")


# ── Inference ────────────────────────────────────────────────────────
def is_wav2lip_installed() -> bool:
    """Cheap pre-flight check used by the health endpoint and to fail
    fast with a clear message rather than a cryptic subprocess error."""
    return WAV2LIP_DIR.is_dir() and WAV2LIP_CKPT.is_file()


async def _run_ffmpeg_dub(face_path: Path, audio_path: Path, out_path: Path) -> None:
    """Reliable lip-sync fallback when Wav2Lip can't render.

    Plays the avatar MP4 (looping if shorter than the audio) while the
    TTS audio plays as the soundtrack. The mouth movement comes from
    the original recording rather than being computed against the TTS,
    so it's not phoneme-accurate — but the avatar's lips/face *do*
    move while the AI speaks, which is what the user actually wants
    visually. Crucially this path has no ML dependencies (only ffmpeg,
    which is required for Wav2Lip too) so it works in every environment
    where Wav2Lip's Python deps are flaky.

    The duration is driven by the audio (-shortest) so the talk video
    ends exactly when the AI stops speaking — same property as the
    silence-trimmed Wav2Lip output, no trailing animation.
    """
    if shutil.which("ffmpeg") is None:
        raise RuntimeError("ffmpeg is required for the video fallback. Install it on the server.")

    cmd = [
        "ffmpeg", "-y",
        "-stream_loop", "-1",            # loop avatar so audio drives length
        "-i", str(face_path),
        "-i", str(audio_path),
        "-map", "0:v:0", "-map", "1:a:0",
        "-c:v", "libx264", "-preset", "veryfast", "-pix_fmt", "yuv420p",
        "-c:a", "aac", "-b:a", "128k",
        "-shortest",                     # cut at end of TTS audio
        "-movflags", "+faststart",       # progressive playback
        str(out_path),
    ]
    log.info("[video_engine] ffmpeg dub fallback: %s", " ".join(cmd))
    proc = await asyncio.create_subprocess_exec(
        *cmd,
        stdout=asyncio.subprocess.PIPE,
        stderr=asyncio.subprocess.PIPE,
    )
    _, stderr = await proc.communicate()
    if proc.returncode != 0:
        err_tail = (stderr or b"").decode("utf-8", "ignore")[-2000:]
        log.error("[video_engine] ffmpeg dub failed rc=%s stderr_tail=\n%s",
                  proc.returncode, err_tail)
        raise RuntimeError(f"ffmpeg dub failed (exit {proc.returncode}).")
    if not out_path.is_file() or out_path.stat().st_size == 0:
        raise RuntimeError("ffmpeg dub produced no output file.")


async def _run_wav2lip(face_path: Path, audio_path: Path, out_path: Path) -> None:
    """Invoke Wav2Lip's inference.py as a subprocess. Blocks an event-loop
    thread for the duration of inference (3–8 s on GPU, 30–90 s on CPU)
    — the caller is expected to hold ``_inference_lock`` so other turns
    wait their turn.

    Three sync-quality fixes vs the previous build:

      1. Audio is RESAMPLED to 16 kHz mono before Wav2Lip sees it.
         The mel encoder was trained on 16 kHz; feeding 24 kHz
         shifted the phoneme→viseme mapping by ~25 ms which read as
         the avatar's mouth running a beat ahead of the audio.

      2. GPU acceleration when CUDA is available (env var
         WAV2LIP_USE_GPU=true or auto-detected). On GPU we also
         raise the wav2lip and face_det batches — both single-digit
         on CPU, double-digit on GPU.

      3. Tighter --pads. The previous 5/12/0/0 was tuned for one
         specific avatar; 0/15/0/0 works better across the full
         32-avatar set because the larger bottom pad covers the
         chin region across different framings without the top pad
         pulling the mask into the nose.
    """
    if not is_wav2lip_installed():
        raise RuntimeError(
            "Wav2Lip is not installed. Run scripts/setup_wav2lip.sh on the server."
        )

    # ── Step 1: resample to 16 kHz ───────────────────────────────────
    # We write the resampled file next to the original so cleanup in
    # get_or_generate() picks it up via the same .wav extension.
    resampled_path = audio_path.with_suffix(".16k.wav")
    try:
        await _resample_wav_to_16k(audio_path, resampled_path)
        audio_path = resampled_path
    except Exception as exc:  # noqa: BLE001
        # If resampling fails we fall back to the original — Wav2Lip
        # will still produce video, just with the historical sync
        # quality.
        log.warning("[video_engine] audio resample failed (%s) — using original sample rate", exc)

    # ── Step 2: choose CPU vs GPU ────────────────────────────────────
    use_gpu = _should_use_wav2lip_gpu()
    wav_batch  = 64 if use_gpu else 4   # mouth-image batch
    face_batch = 16 if use_gpu else 4   # face-detection batch

    cmd = [
        "python", "inference.py",
        "--checkpoint_path", str(WAV2LIP_CKPT),
        "--face",            str(face_path),
        "--audio",           str(audio_path),
        "--outfile",         str(out_path),
        "--pads",            "0", "15", "0", "0",
        "--resize_factor",   "1",
        "--wav2lip_batch_size", str(wav_batch),
        "--face_det_batch_size", str(face_batch),
    ]
    log.info(
        "[video_engine] running wav2lip gpu=%s wav_batch=%d face_batch=%d cmd=%s cwd=%s",
        use_gpu, wav_batch, face_batch, " ".join(cmd), WAV2LIP_DIR,
    )

    # Run inference off the event loop so other WebSocket traffic
    # (text streaming, ping/pong, etc.) stays responsive.
    proc = await asyncio.create_subprocess_exec(
        *cmd,
        cwd=str(WAV2LIP_DIR),
        stdout=asyncio.subprocess.PIPE,
        stderr=asyncio.subprocess.PIPE,
    )
    stdout, stderr = await proc.communicate()

    if proc.returncode != 0:
        # Wav2Lip prints diagnostics to both streams. Surface the tail
        # of stderr — that's where the actual error usually lands —
        # but cap length so we don't bloat the log.
        err_tail = (stderr or b"").decode("utf-8", "ignore")[-2000:]
        log.error("[video_engine] wav2lip failed rc=%s stderr_tail=\n%s",
                  proc.returncode, err_tail)
        raise RuntimeError(f"Wav2Lip inference failed (exit {proc.returncode}). "
                           "See server logs for details.")

    if not out_path.is_file() or out_path.stat().st_size == 0:
        raise RuntimeError("Wav2Lip produced no output file.")


async def get_or_generate(audio_pcm: bytes, avatar_name: str) -> dict:
    """Main entry point.

    Returns a dict::

        {
          "url":     "/static/cache/video/<sha>.mp4",
          "key":     "<sha256>",
          "avatar":  "male_one.mp4",
          "cached":  True | False,
        }

    Caller passes raw 16-bit PCM @ 24 kHz mono (matches what
    `services.openai_service.stream_tts_chunks` produces). We wrap it in
    a WAV file, hash, and either reuse the cached video or generate a
    fresh one. On cache hit the call returns in < 5 ms.
    """
    if not audio_pcm:
        raise ValueError("audio_pcm is empty")
    if not avatar_name:
        avatar_name = pick_avatar()

    avatar_path = AVATAR_DIR / avatar_name
    if not avatar_path.is_file():
        raise FileNotFoundError(
            f"Avatar not found: {avatar_path}. "
            f"Place the 4 MP4 files under static/assets/avatars/."
        )

    # Trim trailing silence BEFORE hashing so two TTS calls with
    # different tail silence still hit the same cache entry — and
    # Wav2Lip animates only the actually-voiced portion of the audio,
    # so the lips stop moving the instant the AI stops talking.
    audio_pcm = trim_trailing_silence(audio_pcm)

    key       = _cache_key(audio_pcm, avatar_name)
    out_path  = _cached_video_path(key)

    if out_path.is_file() and out_path.stat().st_size > 0:
        log.info("[video_engine] cache hit key=%s avatar=%s", key[:12], avatar_name)
        return {"url": cached_url_for(key), "key": key, "avatar": avatar_name, "cached": True}

    log.info("[video_engine] cache miss key=%s avatar=%s — rendering", key[:12], avatar_name)

    # Write the audio to a temp WAV that Wav2Lip can read.
    wav_bytes = pcm_to_wav_bytes(audio_pcm)
    audio_tmp = TMP_DIR / f"{key}.wav"
    audio_tmp.write_bytes(wav_bytes)

    # Render under the global lock so we don't queue up two CPU jobs.
    used_fallback = False
    try:
        async with _inference_lock:
            # Check again inside the lock — another task might have
            # finished generating the same video while we were waiting.
            if out_path.is_file() and out_path.stat().st_size > 0:
                log.info("[video_engine] cache filled while waiting key=%s", key[:12])
            else:
                # Two-tier render strategy:
                #   1. Try Wav2Lip first — gives true phoneme-accurate
                #      lip-sync when its install + weights are healthy.
                #   2. If anything goes wrong (missing pkg_resources,
                #      bad weights, OOM, etc.) fall through to the
                #      ffmpeg dub. The candidate still sees the avatar
                #      visually speaking — just with the recording's
                #      original mouth movement rather than computed
                #      lip-sync. This is the difference between a
                #      working session and a "lip-sync unavailable"
                #      apology.
                try:
                    if is_wav2lip_installed():
                        await _run_wav2lip(avatar_path, audio_tmp, out_path)
                    else:
                        raise RuntimeError("Wav2Lip not installed; using ffmpeg fallback.")
                except Exception as wav_err:
                    log.warning("[video_engine] wav2lip path unavailable (%s) — using ffmpeg dub", wav_err)
                    used_fallback = True
                    await _run_ffmpeg_dub(avatar_path, audio_tmp, out_path)
    finally:
        try: audio_tmp.unlink(missing_ok=True)
        except Exception: pass

    return {
        "url":      cached_url_for(key),
        "key":      key,
        "avatar":   avatar_name,
        "cached":   False,
        "fallback": used_fallback,
    }


def avatar_url(avatar_name: str) -> str:
    """Static URL of an avatar MP4 for direct playback when no
    lip-sync is needed (e.g. as a placeholder while Wav2Lip is
    rendering — the same face that will appear in the video)."""
    return f"/static/assets/avatars/{avatar_name}"


# ── Health surface ──────────────────────────────────────────────────
def health() -> dict:
    """Used by /api/_health/video so ops can verify install + assets
    in one HTTP call from any browser."""
    cache_files = list(CACHE_DIR.glob("*.mp4")) if CACHE_DIR.exists() else []
    return {
        "wav2lip_installed":   is_wav2lip_installed(),
        "wav2lip_dir":         str(WAV2LIP_DIR),
        "checkpoint_present":  WAV2LIP_CKPT.is_file(),
        "ffmpeg_on_path":      shutil.which("ffmpeg") is not None,
        "avatars_found":       list_available_avatars(),
        "avatars_expected":    AVATAR_FILENAMES,
        "avatar_dir":          str(AVATAR_DIR),
        "cache_dir":           str(CACHE_DIR),
        "cache_count":         len(cache_files),
        "cache_size_bytes":    sum(p.stat().st_size for p in cache_files),
    }
