Ë
    ç¹j €  ã                  ó<  — U d Z ddlmZ ddlZddlZddlZddlZddlZddlZddl	Z	ddl
Z
ddlZddlmZ ddlmZ  ej                   e«      Z ee«      j)                  «       j*                  j*                  Zedz  dz  Zedz  d	z  Zed
z  dz  dz  Zedz  dz  dz  Zedz  dz  dz  ZeefD ]  Zej;                  dd¬«       Œ g d¢Zded<   i dd“dd“dd“dd“dd“dd“dd“dd“dd“d d“d!d"“d#d"“d$d"“d%d"“d&d"“d'd"“d(d"“d"d"d"d)œ¥Z d*ed+<   d,d-d.œZ!d*ed/<   dLd0„Z"dMd1„Z#dNd2„Z$dOd3„Z%dOd4„Z&dPdQd5„Z' ejP                  «       Z)da*d6ed7<   d8a+dRd9„Z,dSd:„Z-dTd;„Z.dUd<„Z/dVd=„Z0d>Z1d?Z2e1e2d@dAf	 	 	 	 	 	 	 	 	 dWdB„Z3e1e2f	 	 	 	 	 dXdC„Z4dDZ5dYdE„Z6dSdF„Z7dZdG„Z8dZdH„Z9d[dI„Z:dOdJ„Z;d\dK„Z<y)]ué	  
services/video_engine.py â€” Custom Wav2Lip-based lip-sync video engine.

Replaces the prior Tavus integration. The role-play WebSocket calls
`generate_lipsync_video()` once per AI turn (after TTS audio is fully
streamed) to produce an MP4 of one of our local avatars lip-syncing to
the synthesized speech.

â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€
Why subprocess + cache + lock
â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€
* Wav2Lip's `inference.py` is a CLI; calling it as a subprocess avoids
  importing torch / cv2 / librosa into our FastAPI process â€” those deps
  are heavy, slow to import, and would brittle the entire app if the
  install were broken. The route stays loadable even if Wav2Lip is
  missing; it only fails at call time, with a clear error.
* On CPU, a single Wav2Lip render takes 30â€“90 s. Caching by the SHA-256
  of (audio_bytes + avatar) means a candidate hearing the same scripted
  greeting twice waits only on the first turn â€” every replay is instant.
* An asyncio.Lock serialises inference so two concurrent sessions don't
  thrash the CPU. Conversations stay responsive (the second user just
  waits their turn rather than getting a half-speed render).
â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€
Setup
â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€â”€
Run `bash scripts/setup_wav2lip.sh` once on the server. It:
  â€¢ clones Wav2Lip into  vendor/Wav2Lip/
  â€¢ installs the Python deps into the active venv
  â€¢ downloads the pretrained model into  vendor/Wav2Lip/checkpoints/
  â€¢ verifies ffmpeg is on PATH

Place your 4 avatar MP4s in:
  static/assets/avatars/male_one.mp4
  static/assets/avatars/male_two.mp4
  static/assets/avatars/female_one.mp4
  static/assets/avatars/female_two.mp4

Generated videos go in:
  data/cache/video/<sha256>.mp4
and are served at  /static/cache/video/<sha256>.mp4  via the static
mount in main.py.
é    )ÚannotationsN)ÚPath)ÚOptionalÚvendorÚWav2LipÚcheckpointszwav2lip_gan.pthÚstaticÚassetsÚavatarsÚdataÚcacheÚvideoÚtmpT)ÚparentsÚexist_ok)úfemale_one.mp4úfemale_two.mp4úfemale_three.mp4úfemale_four.mp4úfemale_five.mp4úfemale_six.mp4úfemale_seven.mp4úfemale_eight.mp4úfemale_nine.mp4úfemale_ten.mp4úmale_one.mp4úmale_two.mp4úmale_three.mp4úmale_four.mp4úmale_five.mp4úmale_six.mp4úmale_seven.mp4úmale_eight.mp4úmale_nine.mp4úmale_ten.mp4ú	list[str]ÚAVATAR_FILENAMESr   Úfemaler   r   r   r   r   r   r   r   r   r   Úmaler   r   r   r    r!   r"   )r#   r$   r%   zdict[str, str]ÚAVATAR_GENDERÚonyxÚnova)r)   r(   ÚVOICE_BY_GENDERc                 óÜ   — t         j                  «       r)t        d„ t         j                  d«      D «       «      } | r| S t        D �cg c]  }t         |z  j                  «       r|‘Œ c}S c c}w )ul  Return the names of avatar files actually present on disk.

    Discovery is now DYNAMIC â€” we scan static/assets/avatars/ for any
    .mp4 file rather than relying on the hardcoded AVATAR_FILENAMES.
    That means an operator can drop new avatar clips into the folder
    and they show up in the picker / random pool without a code
    change (per the "scalable so new avatar videos automatically
    appear" requirement on the avatar-selection feature).

    The hardcoded list is still used as a fall-through if the
    directory is missing or empty, so dev environments without the
    asset bundle still boot.
    c              3  óV   K  — | ]!  }|j                  «       sŒ|j                  –— Œ# y ­w©N)Úis_fileÚname©Ú.0Úps     úp/Users/priyanka/Documents/AI_Agent_Assessment/pepole_hub_candidate/people_hub_candidate/services/video_engine.pyÚ	<genexpr>z)list_available_avatars.<locals>.<genexpr>Ç   s   è ø€ ÐOÑ'? !À1Ç9Á9Å;�q—v•vÑ'?ùs   ‚)˜)ú*.mp4)Ú
AVATAR_DIRÚis_dirÚsortedÚglobr'   r1   )Úfoundr2   s     r6   Úlist_available_avatarsr>   ¸   sg   € ô ×ÑÔÜÑO¤z§¡°wÔ'?ÓOÓOˆÙØˆLå)óÙ)�Ü˜Ñ×&Ñ&Ô(ò 	Ð)ñð ùò s   Á A)c                ó”   — | sy| j                  «       }| t        v r	t        |    S |j                  d«      ry|j                  d«      ryy)z×Infer a gender label from a filename. Conventional names are
    `male_*.mp4` / `female_*.mp4`. Anything else falls back to "male"
    as a conservative default (same fallback as AVATAR_GENDER for
    unknown keys).r)   r(   )Úlowerr*   Ú
startswith)r2   r@   s     r6   Ú_gender_from_filenamerB   Ð   sO   € ñ
 ØØ�J‰J‹L€Eð Œ}ÑÜ˜TÑ"Ð"Ø×Ñ˜Ô!ØØ×Ñ˜ÔØØó    c            
     ó  — g } t        «       D ]d  }t        |z  }	 |j                  «       r|j                  «       j                  nd}| j                  |t        |«      |d|› �t        |«      dœ«       Œf | S # t
        $ r d}Y Œ=w xY w)zûReturn one dict per available avatar:
        { name, gender, size_bytes, url }
    Used by the GET /api/avatars endpoint that powers the candidate-
    facing avatar picker modal. Names are sorted so the picker order
    is stable between calls.
    r   ú/static/assets/avatars/)r2   ÚgenderÚ
size_bytesÚurlÚvoice)	r>   r9   r1   ÚstatÚst_sizeÚ	ExceptionÚappendrB   Úvoice_for_avatar)Úitemsr2   ÚpathÚsizes       r6   Úavatar_manifestrR   ã   s�   € ð €EÜ&Ö(ˆÜ˜DÑ ˆð	Ø*.¯,©,¬.�4—9‘9“;×&Ò&¸aˆDð 	�‰ØÜ/°Ó5ØØ3°D°6Ð:Ü*¨4Ó0ñ
õ 	ð )ð €Løô ò 	ØŠDð	ús   š,A6Á6BÂBc                ó.   — t         j                  | d«      S )ui  Return 'male' or 'female' for a given avatar filename.

    Falls back to 'male' for unknown filenames â€” see the comment on
    AVATAR_GENDER. The fallback is conservative on purpose: a wrong
    gender produces the bug we're trying to fix, so we'd rather have
    an unknown avatar default to a single consistent voice than have
    it picked at random.
    r)   )r*   Úget©Úavatar_names    r6   Úavatar_genderrW   û   s   € ô ×Ñ˜[¨&Ó1Ð1rC   c                óN   — t         j                  t        | «      t         d   «      S )a  Return the OpenAI TTS voice ID that matches this avatar's
    gender. The role-play WebSocket calls this immediately after
    picking the avatar and threads the result through every
    stream_tts_chunks() call for the session, so audio and video
    always agree on gender.r)   )r-   rT   rW   rU   s    r6   rN   rN     s!   € ô ×Ñœ}¨[Ó9¼?È6Ñ;RÓSÐSrC   c                ó4  — t        «       xs t        }|rA|j                  «       j                  «       }|D �cg c]  }t	        |«      |k(  sŒ|‘Œ }}|s|}n|}| r&t        j                  | «      }|j                  |«      S t        j                  |«      S c c}w )uM  Return a random avatar filename (e.g. 'male_one.mp4').

    Pass ``seed`` (e.g. the session_id) to get a deterministic pick for
    a given session â€” the same session always uses the same avatar so
    the candidate doesn't see the face change mid-conversation. Calling
    again with the same seed returns the same avatar.

    Pass ``gender`` ('male' or 'female') to restrict the pool to
    avatars of that gender. When omitted the full pool is used. This
    is the seam used by scenarios that want to lock the interviewer's
    gender (e.g. ``ai_gender`` on the scenario JSON).
    )r>   r'   Ústripr@   rW   ÚrandomÚRandomÚchoice)ÚseedrF   Ú	availableÚaÚpoolÚrngs         r6   Úpick_avatarrc     s�   € ô 'Ó(Ò<Ô,<€IÙØ—‘“×%Ñ%Ó'ˆÙ$ÓC™9�a¬°aÓ(8¸FÓ(B’˜9ˆÐCñ Ø‰DàˆÙÜ�m‰m˜DÓ!ˆØ�z‰z˜$ÓÐÜ�=‰=˜ÓÐùò Ds   ·BÁBzOptional[bool]Ú_GPU_CACHEDFc                 óx  — t         ryda t        «       st        j                  d«       y	 dt        › d�} t        j                  dd| gt        t        «      t
        j                  t
        j                  ¬«       t        j                  d	«       y# t        $ r }t        j                  d
|«       Y d}~yd}~ww xY w)u½   Fire-and-forget warmup. Safe to call from a sync context (e.g.
    FastAPI lifespan); we kick off the subprocess and don't wait for
    it. Idempotent â€” only the first call does anything.NTu7   [video_engine] warmup skipped â€” Wav2Lip not installedz$import sys, os
sys.path.insert(0, r'zÉ')
try:
  import torch, cv2, librosa, numpy
  import face_detection  # type: ignore
  print('[video_engine] warmup imports OK')
except Exception as e:
  print('[video_engine] warmup import error:', e)
Úpythonz-c©ÚcwdÚstdoutÚstderrz1[video_engine] Wav2Lip warmup subprocess launchedz*[video_engine] warmup failed (ignored): %s)Ú_WARMUP_STARTEDÚis_wav2lip_installedÚlogÚinfoÚWAV2LIP_DIRÚ
subprocessÚPopenÚstrÚDEVNULLrL   Úwarning)ÚscriptÚexcs     r6   Úwarmup_wav2lip_asyncrw   H  s­   € õ
 ØØ€OÜÔ!Ü�‰ÐJÔKØðGð$Ü$/ =ð 1BðBð 	ô 	×ÑØ�t˜VÐ$Ü”KÓ Ü×%Ñ%Ü×%Ñ%õ		
ô 	�‰ÐDÕEøÜò GÜ�‰Ð@À#×FÑFûðGús   «A$B Â	B9ÂB4Â4B9c                 ó$  — t         �t         S t        j                  dd«      xs dj                  «       j	                  «       } | dv rda y| dv rda yt        j                  d«      €da y	 t        j                  ddgt        j                  t        j                  d	d¬
«      }|j                  dk(  xr d|j                  xs dv a t         rt        j                  d«       t         S t        j                  d«       t         S # t        $ r da Y ŒIw xY w)NÚWAV2LIP_USE_GPUÚ )Ú1ÚtrueÚyesÚonT)Ú0ÚfalseÚnoÚoffFz
nvidia-smiz-Lg       @)ri   rj   ÚtimeoutÚcheckr   s   GPUrC   u=   [video_engine] CUDA GPU detected â€” Wav2Lip will run on GPU.u@   [video_engine] No CUDA GPU detected â€” Wav2Lip will run on CPU.)rd   ÚosÚgetenvrZ   r@   ÚshutilÚwhichrp   ÚrunÚPIPEÚ
returncoderi   rL   rm   rn   )ÚenvÚresults     r6   Ú_should_use_wav2lip_gpurŽ   o  s
  € äÐÜÐä�9‰9Ð&¨Ó+Ò1¨r×
8Ñ
8Ó
:×
@Ñ
@Ó
B€CØ
Ð(Ñ(ØˆØØ
Ð)Ñ)ØˆØô ‡|�|�LÓ!Ð)ØˆØð
Ü—‘Ø˜4Ð Ü—?‘?Ü—?‘?ØØô
ˆð ×(Ñ(¨AÑ-ÒR°&¸V¿]¹]Ò=QÈcÐ2Rˆõ Ü�‰ÐPÔQô Ðô 	�‰ÐSÔTÜÐøô ò ØŠðús   Á*AD ÄDÄDc                óÎ   — t        j                  «       }|j                  | «       |j                  d«       |j                  |j                  d«      «       |j	                  «       S )uD  Hash the audio + avatar pair. Identical (audio, avatar) â†’ same key
    â†’ reuse the cached MP4. We hash the raw audio rather than the prompt
    text because TTS is not strictly deterministic (voice / prosody
    settings can drift) and we want the cache to reflect bit-identical
    output, not best-effort text matches.s   ::úutf-8)ÚhashlibÚsha256ÚupdateÚencodeÚ	hexdigest)Úaudio_bytesrV   Úhs      r6   Ú
_cache_keyr˜   “  sJ   € ô 	�‰Ó€AØ‡H�Hˆ[ÔØ‡H�HˆU„OØ‡H�Hˆ[×Ñ Ó(Ô)Ø�;‰;‹=ÐrC   c                ó   — t         | › d�z  S )Nú.mp4)Ú	CACHE_DIR©Úkeys    r6   Ú_cached_video_pathrž      s   € Ü˜#˜˜d�|Ñ#Ð#rC   c                ó   — d| › d�S )zXURL the frontend should load. Served by the /static/cache mount
    declared in main.py.z/static/cache/video/rš   © rœ   s    r6   Úcached_url_forr¡   ¤  s   € ð " #  dÐ+Ð+rC   iÀ]  é   iô  éP   c                ó`  — | s| S t        | «      |z  }|dk(  r| S d}t        |dz
  dd«      D ]7  }||z  }t        j                  | |||z    dd¬«      }	t	        |	«      |k\  sŒ5|} n |dk  r| S t        |dz  «      }
t        ||dz   |
z   «      }t        |t        ||z  dz  «      «      }| d	||z   S )
u  Strip silent samples from the END of a PCM-16-LE buffer.

    Why this matters: TTS often appends 100â€“400 ms of low-amplitude
    tail. Wav2Lip animates a mouth shape for every audio frame, so
    that tail makes the avatar's lips keep moving for a beat after
    the voice has stopped â€” the exact "trailing animation" the user
    flagged. Trimming the silence before render makes the video end
    on the last spoken syllable.

    `threshold`   â€” amplitude below which a sample counts as silence
                    (16-bit PCM, so range is Â±32 768; 500 â‰ˆ -36 dBFS).
    `min_keep_ms` â€” never trim below this length even if everything
                    looks silent (defensive â€” keeps the engine sane
                    on extremely short utterances).
    r   éÿÿÿÿé   ÚlittleT)Úsignedg{®Gáz¤?iè  N)ÚlenÚrangeÚintÚ
from_bytesÚabsÚminÚmax)ÚpcmÚsample_rateÚsample_widthÚ	thresholdÚmin_keep_msÚsamplesÚlast_voicedÚir‚   ÚvalÚpad_samplesÚkeep_sampless               r6   Útrim_trailing_silencer»   ¯  sä   € ñ( Øˆ
Ü�#‹h˜,Ñ&€GØ�!‚|Øˆ
ð €KÜ�7˜Q‘;  BÖ'ˆØ�,Ñˆä�n‰n˜S  S¨<Ñ%7Ð8¸(È4ˆnÓPˆÜˆs‹8�yÓ ØˆKÙð (ð �Q‚Øˆ
ô �k DÑ(Ó)€KÜ�w ¨a¡°+Ñ =Ó>€LÜ�|¤S¨°{Ñ)BÀTÑ)IÓ%JÓK€LØÐ,� Ñ,Ð-Ð-rC   c                óF  — ddl }|j                  «       }t        j                  |d«      5 }|j	                  d«       |j                  |«       |j                  |«       |j                  | «       ddd«       |j                  «       S # 1 sw Y   |j                  «       S xY w)u›   Wrap raw PCM-16-LE bytes in a WAV header. Wav2Lip's audio loader
    (librosa) needs a real audio container â€” it can't parse headerless
    PCM directly.r   NÚwbr¦   )	ÚioÚBytesIOÚwaveÚopenÚsetnchannelsÚsetsampwidthÚsetframerateÚwriteframesÚgetvalue)r°   r±   r²   Ú_ioÚbufÚwfs         r6   Úpcm_to_wav_bytesrÊ   Ý  sy   € ó Ø
�+‰+‹-€CÜ	�‰�3˜Ô	 Ø
�‰˜ÔØ
�‰˜Ô%Ø
�‰˜Ô$Ø
�‰�sÔ÷	 
ð
 �<‰<‹>Ð÷ 
ð
 �<‰<‹>Ðús   «ABÂB i€>  c              ƒ  ó¶  K  — t        j                  d«      €t        d«      ‚dddt        | «      dddt        t        «      d	d
ddt        |«      g}t        j                  |t
        j                  j                  t
        j                  j                  dœŽƒ d{  –—† }|j                  «       ƒ d{  –—† \  }}|j                  dk7  rS|xs dj                  dd«      dd }t        j                  d|j                  |«       t        d|j                  › d�«      ‚|j                  «       r|j                  «       j                   dk(  rt        d«      ‚y7 Œº7 Œ¤­w)aS  Resample any WAV/PCM file to 16 kHz mono PCM-16-LE. No-op if the
    input is already 16 kHz mono; ffmpeg is smart enough to copy in
    that case. We always go through ffmpeg rather than librosa here
    because Wav2Lip's own dependency tree already pulls in ffmpeg
    and we don't want to add another resampler that might disagree.
    ÚffmpegNz0ffmpeg is required for Wav2Lip audio resampling.ú-yú-iz-acr{   z-arú-c:aÚ	pcm_s16lez	-loglevelÚerror©ri   rj   r   rC   r�   Úignorei$úÿÿz6[video_engine] ffmpeg resample failed rc=%s stderr=
%szffmpeg resample failed (exit ú).z(ffmpeg resample produced no output file.)r‡   rˆ   ÚRuntimeErrorrr   ÚWAV2LIP_NATIVE_SAMPLE_RATEÚasyncioÚcreate_subprocess_execrp   rŠ   Úcommunicater‹   Údecoderm   rÑ   r1   rJ   rK   )Úsrc_pathÚdst_pathÚcmdÚprocÚ_rj   Úerr_tails          r6   Ú_resample_wav_to_16krá   ø  s>  è ø€ ô ‡|�|�HÓÐ%ÜÐMÓNÐNà�$ØŒs�8‹}ØˆsØŒsÔ-Ó.Ø�Ø�WÜˆH‹ð€Cô ×/Ñ/Ø	Ü×!Ñ!×&Ñ&Ü×!Ñ!×&Ñ&ò÷ €Dð
 ×&Ñ&Ó(×(�I€A€vØ‡�˜!ÒØ’M˜c×)Ñ)¨'°8Ó<¸U¸VÐDˆÜ�	‰	ÐKØ—/‘/ 8ô	-äÐ:¸4¿?¹?Ð:KÈ2ÐNÓOÐOØ×ÑÔ §¡£×!8Ñ!8¸AÒ!=ÜÐEÓFÐFð ">ðøð
 )ús%   ‚BEÂEÂEÂ2EÂ3B#EÅEc                 óV   — t         j                  «       xr t        j                  «       S )z„Cheap pre-flight check used by the health endpoint and to fail
    fast with a clear message rather than a cryptic subprocess error.)ro   r:   ÚWAV2LIP_CKPTr1   r    rC   r6   rl   rl     s!   € ô ×ÑÓÒ:¤L×$8Ñ$8Ó$:Ð:rC   c              ƒ  ó  K  — t        j                  d«      €t        d«      ‚dddddt        | «      dt        |«      dd	dd
dddddddddddddt        |«      g}t        j                  ddj                  |«      «       t        j                  |t        j                  j                  t        j                  j                  dœŽƒ d{  –—† }|j                  «       ƒ d{  –—† \  }}|j                  dk7  rS|xs dj                  dd«      dd }t        j                  d |j                  |«       t        d!|j                  › d"�«      ‚|j                  «       r|j!                  «       j"                  dk(  rt        d#«      ‚y7 Œº7 Œ¤­w)$u  Reliable lip-sync fallback when Wav2Lip can't render.

    Plays the avatar MP4 (looping if shorter than the audio) while the
    TTS audio plays as the soundtrack. The mouth movement comes from
    the original recording rather than being computed against the TTS,
    so it's not phoneme-accurate â€” but the avatar's lips/face *do*
    move while the AI speaks, which is what the user actually wants
    visually. Crucially this path has no ML dependencies (only ffmpeg,
    which is required for Wav2Lip too) so it works in every environment
    where Wav2Lip's Python deps are flaky.

    The duration is driven by the audio (-shortest) so the talk video
    ends exactly when the AI stops speaking â€” same property as the
    silence-trimmed Wav2Lip output, no trailing animation.
    rÌ   NzDffmpeg is required for the video fallback. Install it on the server.rÍ   z-stream_loopz-1rÎ   z-mapz0:v:0z1:a:0z-c:vÚlibx264z-presetÚveryfastz-pix_fmtÚyuv420prÏ   Úaacz-b:aÚ128kz	-shortestz	-movflagsz
+faststartz&[video_engine] ffmpeg dub fallback: %sÚ rÒ   r   rC   r�   rÓ   é0øÿÿz6[video_engine] ffmpeg dub failed rc=%s stderr_tail=
%szffmpeg dub failed (exit rÔ   z#ffmpeg dub produced no output file.)r‡   rˆ   rÕ   rr   rm   rn   Újoinr×   rØ   rp   rŠ   rÙ   r‹   rÚ   rÑ   r1   rJ   rK   )Ú	face_pathÚ
audio_pathÚout_pathrÝ   rÞ   rß   rj   rà   s           r6   Ú_run_ffmpeg_dubrð      sr  è ø€ ô  ‡|�|�HÓÐ%ÜÐaÓbÐbð 	�$Ø˜ØŒc�)‹nØŒc�*‹oØ�˜ Ø�	˜9 j°*¸iØ��v˜vØØ�\ÜˆH‹ð€Cô ‡H�HÐ5°s·x±xÀ³}ÔEÜ×/Ñ/Ø	Ü×!Ñ!×&Ñ&Ü×!Ñ!×&Ñ&ò÷ €Dð
 ×&Ñ&Ó(×(�I€A€vØ‡�˜!ÒØ’M˜c×)Ñ)¨'°8Ó<¸U¸VÐDˆÜ�	‰	ÐKØ—/‘/ 8ô	-äÐ5°d·o±oÐ5FÀbÐIÓJÐJØ×ÑÔ §¡£×!8Ñ!8¸AÒ!=ÜÐ@ÓAÐAð ">ðøð
 )ús%   ‚CFÃFÃ	FÃ FÃ!B#FÆFc              ƒ  ó8  K  — t        «       st        d«      ‚|j                  d«      }	 t        ||«      ƒ d{  –—†  |}t        «       }|rdnd}|rdnd}dd	d
t        t        «      dt        | «      dt        |«      dt        |«      ddddddddt        |«      dt        |«      g}t
        j                  d|||dj                  |«      t        «       t        j                  |t        t        «      t        j                  j                   t        j                  j                   dœŽƒ d{  –—† }	|	j#                  «       ƒ d{  –—† \  }
}|	j$                  dk7  rS|xs dj'                  dd«      dd }t
        j)                  d|	j$                  |«       t        d|	j$                  › d�«      ‚|j+                  «       r|j-                  «       j.                  dk(  rt        d «      ‚y7 �Œ­# t        $ r!}t
        j                  d|«       Y d}~�ŒÎd}~ww xY w7 Œê7 ŒÔ­w)!uE  Invoke Wav2Lip's inference.py as a subprocess. Blocks an event-loop
    thread for the duration of inference (3â€“8 s on GPU, 30â€“90 s on CPU)
    â€” the caller is expected to hold ``_inference_lock`` so other turns
    wait their turn.

    Three sync-quality fixes vs the previous build:

      1. Audio is RESAMPLED to 16 kHz mono before Wav2Lip sees it.
         The mel encoder was trained on 16 kHz; feeding 24 kHz
         shifted the phonemeâ†’viseme mapping by ~25 ms which read as
         the avatar's mouth running a beat ahead of the audio.

      2. GPU acceleration when CUDA is available (env var
         WAV2LIP_USE_GPU=true or auto-detected). On GPU we also
         raise the wav2lip and face_det batches â€” both single-digit
         on CPU, double-digit on GPU.

      3. Tighter --pads. The previous 5/12/0/0 was tuned for one
         specific avatar; 0/15/0/0 works better across the full
         32-avatar set because the larger bottom pad covers the
         chin region across different framings without the top pad
         pulling the mask into the nose.
    zEWav2Lip is not installed. Run scripts/setup_wav2lip.sh on the server.z.16k.wavNuH   [video_engine] audio resample failed (%s) â€” using original sample rateé@   é   é   rf   zinference.pyz--checkpoint_pathz--facez--audioz	--outfilez--padsr   Ú15z--resize_factorr{   z--wav2lip_batch_sizez--face_det_batch_sizezN[video_engine] running wav2lip gpu=%s wav_batch=%d face_batch=%d cmd=%s cwd=%srê   rg   r   rC   r�   rÓ   rë   z3[video_engine] wav2lip failed rc=%s stderr_tail=
%szWav2Lip inference failed (exit z). See server logs for details.z Wav2Lip produced no output file.)rl   rÕ   Úwith_suffixrá   rL   rm   rt   rŽ   rr   rã   rn   rì   ro   r×   rØ   rp   rŠ   rÙ   r‹   rÚ   rÑ   r1   rJ   rK   )rí   rî   rï   Úresampled_pathrv   Úuse_gpuÚ	wav_batchÚ
face_batchrÝ   rÞ   ri   rj   rà   s                r6   Ú_run_wav2liprû   O  s  è ø€ ô0  Ô!ÜØSó
ð 	
ð  ×+Ñ+¨JÓ7€NðeÜ" :¨~Ó>×>Ð>Ø#ˆ
ô &Ó'€GÙ‘ A€IÙ‘ A€Jð 	�.ØœS¤Ó.ØœS ›^ØœS ›_ØœS ›]Ø˜S $¨¨SØ˜SØ¤ I£Ø¤ Z£ð
€Cô ‡H�HØXØ�˜J¨¯©°«´{ôô ×/Ñ/Ø	Ü”ÓÜ×!Ñ!×&Ñ&Ü×!Ñ!×&Ñ&ò	÷ €Dð  ×+Ñ+Ó-×-�N€FˆFà‡�˜!Òð ’M˜c×)Ñ)¨'°8Ó<¸U¸VÐDˆÜ�	‰	ÐHØ—/‘/ 8ô	-äÐ<¸T¿_¹_Ð<Mð N:ð :ó ;ð 	;ð ×ÑÔ §¡£×!8Ñ!8¸AÒ!=ÜÐ=Ó>Ð>ð ">ðc 	?úäò eô 	�‰Ð^Ð`c×dÒdûð	eúð:øð .ús^   ‚'HªG) ¹G&ºG) Á C+HÄ+HÄ,HÅHÅB"HÇ&G) Ç)	HÇ2HÈHÈHÈHÈHc              ƒ  óœ  K  — | st        d«      ‚|s
t        «       }t        |z  }|j                  «       st	        d|› d�«      ‚t        | «      } t        | |«      }t        |«      }|j                  «       rG|j                  «       j                  dkD  r*t        j                  d|dd |«       t        |«      ||dd	œS t        j                  d
|dd |«       t        | «      }t        |› d�z  }|j                  |«       d}	 t         4 ƒd{  –—†  |j                  «       r7|j                  «       j                  dkD  rt        j                  d|dd «       n-	 t#        «       rt%        |||«      ƒ d{  –—†  nt'        d«      ‚	 ddd«      ƒd{  –—†  |j/                  d¬«       t        |«      ||d|dœS 7 Œ«7 ŒF# t(        $ r8}t        j+                  d|«       d}t-        |||«      ƒ d{  –—†7   Y d}~Œsd}~ww xY w7 Œo# 1 ƒd{  –—†7  sw Y   ŒxY w# t(        $ r Y Œ{w xY w# |j/                  d¬«       w # t(        $ r Y w w xY wxY w­w)aå  Main entry point.

    Returns a dict::

        {
          "url":     "/static/cache/video/<sha>.mp4",
          "key":     "<sha256>",
          "avatar":  "male_one.mp4",
          "cached":  True | False,
        }

    Caller passes raw 16-bit PCM @ 24 kHz mono (matches what
    `services.openai_service.stream_tts_chunks` produces). We wrap it in
    a WAV file, hash, and either reuse the cached video or generate a
    fresh one. On cache hit the call returns in < 5 ms.
    zaudio_pcm is emptyzAvatar not found: z5. Place the 4 MP4 files under static/assets/avatars/.r   z)[video_engine] cache hit key=%s avatar=%sNé   T)rH   r�   ÚavatarÚcachedu8   [video_engine] cache miss key=%s avatar=%s â€” renderingz.wavFz0[video_engine] cache filled while waiting key=%sz-Wav2Lip not installed; using ffmpeg fallback.uA   [video_engine] wav2lip path unavailable (%s) â€” using ffmpeg dub)Ú
missing_ok)rH   r�   rþ   rÿ   Úfallback)Ú
ValueErrorrc   r9   r1   ÚFileNotFoundErrorr»   r˜   rž   rJ   rK   rm   rn   r¡   rÊ   ÚTMP_DIRÚwrite_bytesÚ_inference_lockrl   rû   rÕ   rL   rt   rð   Úunlink)	Ú	audio_pcmrV   Úavatar_pathr�   rï   Ú	wav_bytesÚ	audio_tmpÚused_fallbackÚwav_errs	            r6   Úget_or_generater  ¦  s:  è ø€ ñ" ÜÐ-Ó.Ð.ÙÜ!“mˆä˜{Ñ*€KØ×ÑÔ ÜØ   ð .Bð Có
ð 	
ô & iÓ0€Iä˜9 kÓ2€CÜ" 3Ó'€Hà×ÑÔ˜hŸm™m›o×5Ñ5¸Ò9Ü�‰Ð<¸cÀ#À2¸hÈÔTÜ% cÓ*°3À+ÐY]Ñ^Ð^ä‡H�HÐGÈÈSÈbÈÐS^Ô_ô ! Ó+€IÜ˜S˜E ˜,Ñ&€IØ×Ñ˜)Ô$ð €Mðß"–?ð ×ÑÔ! h§m¡m£o×&=Ñ&=ÀÒ&AÜ—‘ÐKÈSÐQTÐRTÈXÕVðLÜ+Ô-Ü*¨;¸	À8ÓL×LÑLä*Ð+ZÓ[Ð[ð M÷' #—?ð6 ×Ñ¨ÐÔ.ô # 3Ó'ØØØØ!ñð ð= #øð& Mùô !ò LÜ—K‘KÐ cÐelÔmØ$(�MÜ)¨+°yÀ(ÓK×KÖKûðLúð- #ø—?—?‘?ûô8 Ò™$Ðûð ×Ñ¨ÐÕ.øÜÒ™$Ðÿsñ   ‚C?IÄ
H& ÄF8ÄH& ÄAHÅF<Å3F:Å4F<ÆHÆH& ÆH ÆH& ÆH Æ'IÆ8H& Æ:F<Æ<	G=Ç(G8Ç-G0Ç.G8Ç3HÇ8G=Ç=HÈ H& ÈHÈHÈ	HÈH& È	H#È IÈ"H#È#IÈ&I	È'H:È9I	È:	IÉI	ÉIÉI	É	Ic                ó   — d| › �S )u¶   Static URL of an avatar MP4 for direct playback when no
    lip-sync is needed (e.g. as a placeholder while Wav2Lip is
    rendering â€” the same face that will appear in the video).rE   r    rU   s    r6   Ú
avatar_urlr    s   € ð % [ MÐ2Ð2rC   c                 óz  — t         j                  «       rt        t         j                  d«      «      ng } t	        «       t        t        «      t        j                  «       t        j                  d«      dut        «       t        t        t        «      t        t         «      t        | «      t        d„ | D «       «      dœ
S )zdUsed by /api/_health/video so ops can verify install + assets
    in one HTTP call from any browser.r8   rÌ   Nc              3  óP   K  — | ]  }|j                  «       j                  –— Œ  y ­wr0   )rJ   rK   r3   s     r6   r7   zhealth.<locals>.<genexpr>  s   è ø€ Ð"I¹[¸ 1§6¡6£8×#3Õ#3¹[ùs   ‚$&)
Úwav2lip_installedÚwav2lip_dirÚcheckpoint_presentÚffmpeg_on_pathÚavatars_foundÚavatars_expectedÚ
avatar_dirÚ	cache_dirÚcache_countÚcache_size_bytes)r›   ÚexistsÚlistr<   rl   rr   ro   rã   r1   r‡   rˆ   r>   r'   r9   r©   Úsum)Úcache_filess    r6   Úhealthr!  	  s†   € ô 4=×3CÑ3CÔ3E”$”y—~‘~ gÓ.Ô/È2€Kä3Ó5Ü"¤;Ó/Ü+×3Ñ3Ó5Ü%Ÿ|™|¨HÓ5¸TÐAÜ5Ó7Ü/Ü"¤:›Ü"¤9›~Ü" ;Ó/Ü"Ñ"I¹[Ó"IÓIñð rC   )Úreturnr&   )r2   rr   r"  rr   )r"  z
list[dict])rV   rr   r"  rr   )NN)r^   úOptional[str]rF   r#  r"  rr   )r"  ÚNone)r"  Úbool)r–   ÚbytesrV   rr   r"  rr   )r�   rr   r"  r   )r�   rr   r"  rr   )r°   r&  r±   r«   r²   r«   r³   r«   r´   r«   r"  r&  )r°   r&  r±   r«   r²   r«   r"  r&  )rÛ   r   rÜ   r   r"  r$  )rí   r   rî   r   rï   r   r"  r$  )r  r&  rV   rr   r"  Údict)r"  r'  )=Ú__doc__Ú
__future__r   r×   r‘   Úloggingr…   r[   r‡   Ústructrp   rÀ   Úpathlibr   Útypingr   Ú	getLoggerÚ__name__rm   Ú__file__ÚresolveÚparentÚBASE_DIRro   rã   r9   r›   r  Ú_dÚmkdirr'   Ú__annotations__r*   r-   r>   rB   rR   rW   rN   rc   ÚLockr  rd   rk   rw   rŽ   r˜   rž   r¡   ÚTTS_SAMPLE_RATEÚTTS_SAMPLE_WIDTHr»   rÊ   rÖ   rá   rl   rð   rû   r  r  r!  r    rC   r6   Ú<module>r:     s÷  ðò)õT #ã Û Û Û 	Û Û Û Û Û Ý Ý à€g×Ñ˜Ó!€ñ �x“.×(Ñ(Ó*×1Ñ1×8Ñ8€Ø˜XÑ%¨	Ñ1€Ø Ñ-Ð0AÑA€Ø˜XÑ%¨Ñ0°9Ñ<€
Ø˜VÑ# gÑ-°Ñ7€	Ø˜VÑ# gÑ-°Ñ5€ð �gÓ
€BØ‡H�H�T D€HÕ)ð ò(Ð �)ó (ð`)!Ø˜ð)!ð ˜ð)!ð
 ˜ð)!ð ˜ð)!ð ˜ð)!ð ˜ð)!ð ˜ð)!ð ˜ð)!ð" ˜ð#)!ð& ˜ð')!ð* �fð+)!ð. �fð/)!ð2 �fð3)!ð6 �fð7)!ð: �fð;)!ð> �fð?)!ðB �fðC)!ðF ààòO)!€ˆ~ó )ðl Øñ#€�ó óó0ó&ó0	2óTôð@ �'—,‘,“.€ð #€ˆ^Ó "ð €ó$GóN óH
ó$ó,ð €ØÐ ð .=Ø.>Ø+.Ø-/ð	+.Ø'*ð+.à(+ð+.ð &)ð+.ð (+ð	+.ð 5:ó	+.ð^ )8Ø)9ðØ"%ðà#&ðà>Cóð0 $Ð óGóD;ó,Bó^T?ónXóv3ôrC   