"""
services/openai_service.py — All OpenAI interactions
=====================================================
Single module for both the synchronous "AI compiler / evaluator" surface
used by the Coding module AND the asynchronous streaming surface used by
the Role Play module (chat, TTS, STT, sentiment, end-of-session report).

Key + base URL come from `config.settings` — never hardcoded.
"""
from __future__ import annotations

import base64
import json
import logging
import os
import re
from io import BytesIO
from typing import Any, AsyncGenerator

from openai import AsyncOpenAI, OpenAI

from config import settings

# Lazy import so missing models don't blow up the module at import time
try:
    from models.schemas import ChatMessage, Scenario  # noqa: F401
except Exception:  # pragma: no cover
    ChatMessage = Any  # type: ignore[assignment, misc]
    Scenario = Any  # type: ignore[assignment, misc]

log = logging.getLogger(__name__)


# ── Compatibility shim: lowercase config accessors ───────────────────────────
# Some legacy code in this module reads `_s.openai_chat_model`, etc.
# We expose those via a thin object that maps to the canonical UPPERCASE
# settings keys, with safe defaults for fields that don't exist in .env.

class _SettingsShim:
    @property
    def openai_api_key(self) -> str:        return settings.OPENAI_API_KEY
    @property
    def openai_base_url(self) -> str:       return settings.OPENAI_BASE_URL
    @property
    def openai_chat_model(self) -> str:     return settings.OPENAI_MODEL
    @property
    def openai_tts_model(self) -> str:      return os.getenv("OPENAI_TTS_MODEL", "tts-1")
    @property
    def openai_tts_voice(self) -> str:      return os.getenv("OPENAI_TTS_VOICE", "alloy")
    @property
    def openai_tts_format(self) -> str:     return os.getenv("OPENAI_TTS_FORMAT", "pcm")
    @property
    def openai_stt_model(self) -> str:      return os.getenv("OPENAI_STT_MODEL", "whisper-1")

_s = _SettingsShim()


# ── Sync client (compile/evaluate/captcha) ───────────────────────────────────

_sync_client: OpenAI | None = None

def get_client() -> OpenAI:
    global _sync_client
    if _sync_client is None:
        _sync_client = OpenAI(
            api_key=settings.OPENAI_API_KEY,
            base_url=settings.OPENAI_BASE_URL,
        )
    return _sync_client


def _async_client() -> AsyncOpenAI:
    """Return a fresh AsyncOpenAI client (cheap; no connection held)."""
    return AsyncOpenAI(
        api_key=settings.OPENAI_API_KEY,
        base_url=settings.OPENAI_BASE_URL,
    )


def _chat(messages: list[dict], max_tokens: int, temperature: float | None = None) -> str:
    resp = get_client().chat.completions.create(
        model=settings.OPENAI_MODEL,
        messages=messages,
        temperature=temperature if temperature is not None else settings.OPENAI_TEMPERATURE,
        max_tokens=max_tokens,
    )
    return resp.choices[0].message.content or ""


def _extract_json(raw: str) -> Any:
    cleaned = re.sub(r"```(?:json)?", "", raw).replace("```", "").strip()
    return json.loads(cleaned)


# ─────────────────────────────────────────────────────────────────────────────
# AI-Compiler (Coding module — Run Code)
# ─────────────────────────────────────────────────────────────────────────────

COMPILER_SYSTEM = """You are a strict compiler and runtime environment. Your ONLY role is to simulate \
compiler/interpreter output for candidate assessment.

ABSOLUTE RULES:
1. DO NOT provide hints, suggestions, fixes, or solutions.
2. DO NOT describe what the code does.
3. DO NOT offer encouragement or commentary.
4. Respond EXACTLY as the real compiler/interpreter would — nothing more.
5. If code has errors: show compiler error messages (file:line:col: error: description).
6. If code runs successfully: output "Program executed successfully." then any stdout.
7. If code is logically wrong but compiles: show runtime output (may be incorrect).

Format: raw compiler output only."""


def compile_code(language: str, code: str, question_title: str) -> dict:
    """Simulate compilation. Returns compiler-style output.

    Failure modes are normalised to candidate-safe messages — never echo
    the upstream OpenAI exception class or HTTP body, which can contain
    rate-limit URLs, request IDs, or organisation slugs.
    """
    if not settings.OPENAI_API_KEY:
        log.error("compile_code: OPENAI_API_KEY is not configured")
        return {
            "output": "Code analysis is temporarily unavailable. "
                      "Your code is saved — please try again shortly.",
            "status": "error",
        }

    user_msg = (
        f"Compile and execute this {language} code for question '{question_title}':\n"
        f"```\n{code}\n```"
    )
    try:
        raw = _chat(
            messages=[
                {"role": "system", "content": COMPILER_SYSTEM},
                {"role": "user",   "content": user_msg},
            ],
            max_tokens=settings.OPENAI_MAX_TOKENS_COMPILE,
            temperature=0.0,
        )
    except Exception as exc:  # noqa: BLE001
        # Log the type but never the full message — exception text from the
        # OpenAI SDK frequently includes URLs, org IDs and request IDs.
        log.warning("compile_code failed: %s", type(exc).__name__)
        return {
            "output": "The AI compiler is temporarily unavailable. "
                      "Your code has not been changed — please try again in a moment.",
            "status": "error",
        }

    is_success = "executed successfully" in (raw or "").lower()
    return {"output": raw or "", "status": "success" if is_success else "error"}


# ─────────────────────────────────────────────────────────────────────────────
# AI Evaluator (Coding module — Submit)
# ─────────────────────────────────────────────────────────────────────────────

EVAL_SYSTEM = """You are an expert code evaluator for a candidate assessment platform.
Evaluate the submitted code objectively. Return ONLY valid JSON, no markdown.

JSON structure:
{
  "result": "correct" | "incorrect" | "partial",
  "score": 0.0 to 1.0,
  "feedback": "2-3 sentence constructive feedback",
  "strengths": ["..."],
  "improvements": ["..."]
}

Rules:
- "correct": code solves the problem fully and efficiently
- "partial": code is on the right track but has bugs or edge case failures
- "incorrect": code does not solve the problem
- Be strict but fair
- Never reveal the correct answer
- Feedback should be educationally useful without giving away the solution"""


def evaluate_code(language: str, code: str, question_text: str, max_marks: int) -> dict:
    """Evaluate submitted code. Returns result, score, feedback, marks."""
    user_msg = (
        f"Question: {question_text}\n\n"
        f"Language: {language}\n\n"
        f"Submitted code:\n```{language}\n{code}\n```"
    )
    try:
        raw = _chat(
            messages=[
                {"role": "system", "content": EVAL_SYSTEM},
                {"role": "user",   "content": user_msg},
            ],
            max_tokens=settings.OPENAI_MAX_TOKENS_EVAL,
            temperature=0.1,
        )
        data = _extract_json(raw)
    except Exception as exc:  # noqa: BLE001
        log.warning("evaluate_code failed: %s", exc)
        data = {
            "result": "incorrect",
            "score": 0.0,
            "feedback": "Evaluation could not be completed. Please try again.",
            "strengths": [],
            "improvements": [],
        }

    score  = float(data.get("score", 0.0) or 0.0)
    result = data.get("result", "incorrect")
    marks  = round(max_marks * score)

    return {
        "ai_result":    result,
        "ai_feedback":  data.get("feedback", ""),
        "ai_score":     score,
        "ai_marks":     marks,
        "max_marks":    max_marks,
        "strengths":    data.get("strengths", []) or [],
        "improvements": data.get("improvements", []) or [],
    }


# ─────────────────────────────────────────────────────────────────────────────
# AI CAPTCHA Generator
# ─────────────────────────────────────────────────────────────────────────────

CAPTCHA_SYSTEM = """Generate a simple visual CAPTCHA challenge for human verification.
Return ONLY valid JSON:
{
  "category": "string (e.g. 'vehicles', 'animals', 'food')",
  "question": "string (e.g. 'Select all images showing vehicles')",
  "items": [
    {"emoji": "🚗", "label": "vehicles", "is_target": true}
  ]
}
Rules:
- Exactly 9 items in a 3×3 grid
- 3–5 items should be the target category
- Items should be emojis only
- Keep it easy enough for any adult in 5 seconds"""


def _captcha_fallback() -> dict:
    return {
        "category": "vehicles",
        "question": "Select all images showing vehicles",
        "items": [
            {"emoji": "🚗", "label": "vehicles", "is_target": True},
            {"emoji": "🌳", "label": "nature",   "is_target": False},
            {"emoji": "🚕", "label": "vehicles", "is_target": True},
            {"emoji": "🍕", "label": "food",     "is_target": False},
            {"emoji": "✈️", "label": "vehicles", "is_target": True},
            {"emoji": "🐶", "label": "animals",  "is_target": False},
            {"emoji": "🚂", "label": "vehicles", "is_target": True},
            {"emoji": "⛵", "label": "vehicles", "is_target": True},
            {"emoji": "🌺", "label": "nature",   "is_target": False},
        ],
    }


def generate_captcha_challenge() -> dict:
    """Generate a visual emoji CAPTCHA via AI, with a deterministic fallback."""
    try:
        raw = _chat(
            messages=[
                {"role": "system", "content": CAPTCHA_SYSTEM},
                {"role": "user",   "content": "Generate a fresh CAPTCHA challenge with a random category."},
            ],
            max_tokens=settings.OPENAI_MAX_TOKENS_CAPTCHA,
            temperature=0.9,
        )
        return _extract_json(raw)
    except Exception as exc:  # noqa: BLE001
        log.info("captcha AI generation failed (%s) — using fallback", exc)
        return _captcha_fallback()


# ─────────────────────────────────────────────────────────────────────────────
# Role Play — System prompt
# ─────────────────────────────────────────────────────────────────────────────

def _build_system_prompt(scenario) -> str:
    """Build a system prompt for a role-play scenario.

    Accepts either a Scenario model instance OR a plain dict (since the
    external Role Play API may return arbitrary JSON shapes — see the
    'build defensively' direction). All field reads are safe."""
    g = (lambda k, default="": (
        getattr(scenario, k, None) if not isinstance(scenario, dict) else scenario.get(k)
    ) or default)

    primary_lang = g('language_name', 'English')
    return (
        f"You are ONLY {g('ai_character', 'the customer')}. "
        f"You are NOT {g('learner_role', 'the candidate')}. "
        f"You must NEVER speak as if you are the candidate or offer solutions as one.\n\n"
        f"SCENARIO CONTEXT:\n{g('context', g('description', 'Practice the conversation realistically.'))}\n\n"
        f"YOUR CHARACTER:\n"
        f"{g('ai_personality', 'Be realistic, emotionally consistent, and challenging.')}\n\n"
        f"CRITICAL RULES — follow these without exception:\n"
        f"1. You are ALWAYS {g('ai_character', 'the customer')}. The learner is ALWAYS {g('learner_role', 'the candidate')}.\n"
        f"2. NEVER switch roles. NEVER give advice as if you were the candidate.\n"
        f"3. React as your character would — emotionally, realistically, in the moment.\n"
        # Multilingual rule (rule 4): respond in the candidate's
        # spoken/written language. The scenario's primary language is
        # only a fallback for when the candidate's language can't be
        # detected (silent turn, transcription empty). This is what
        # lets a candidate switch to Hindi mid-conversation without
        # the AI breaking flow.
        f"4. LANGUAGE: detect the language the candidate is using in their most recent message "
        f"and reply in that same language. Match their script (Devanagari/Roman/etc.) "
        f"and register (formal/casual). If their message is ambiguous, mixed, or empty, "
        f"default to {primary_lang}. Do not announce or comment on the language switch.\n"
        f"5. Keep responses to 2–3 sentences maximum. Be direct and in-character.\n"
        f"6. Do NOT resolve the situation easily — the learner must earn it through skill.\n"
        f"7. Never say you are an AI. Never mention this is training.\n"
        f"8. Difficulty: {g('difficulty', 'Intermediate')}.\n"
        f"9. Session lasts approximately {g('turns', 8)} exchanges total.\n"
    )


# ─────────────────────────────────────────────────────────────────────────────
# Role Play — Streaming chat
# ─────────────────────────────────────────────────────────────────────────────

async def stream_ai_response(messages, scenario) -> AsyncGenerator[str, None]:
    """Stream GPT tokens for the AI character's reply."""
    openai_messages = [{"role": "system", "content": _build_system_prompt(scenario)}]

    for m in messages:
        role = "user" if getattr(m, "role", None) == "user" else "assistant"
        content = getattr(m, "content", "") or ""
        if content in ("[session_start]",):
            continue
        openai_messages.append({"role": role, "content": content})

    try:
        client = _async_client()
        stream = await client.chat.completions.create(
            model=_s.openai_chat_model,
            messages=openai_messages,
            stream=True,
            temperature=0.85,
            max_tokens=300,
        )
        async for chunk in stream:
            delta = chunk.choices[0].delta.content or ""
            if delta:
                yield delta
    except Exception as exc:  # noqa: BLE001
        log.error("[OpenAI] stream_ai_response failed: %s", exc)
        yield f"[I'm having trouble responding. Error: {exc}]"


# ─────────────────────────────────────────────────────────────────────────────
# Role Play — TTS streaming (PCM, zero-lag)
# ─────────────────────────────────────────────────────────────────────────────

async def stream_tts_chunks(
    text: str,
    voice: str | None = None,
) -> AsyncGenerator[str, None]:
    """Stream TTS audio as base64-encoded raw PCM16 chunks (24 kHz mono).

    Pass ``voice`` to override the env-default OPENAI_TTS_VOICE for a
    single call. The role-play WebSocket uses this to pair the voice
    with the chosen avatar's gender (see services/video_engine.py
    `voice_for_avatar`), so a female avatar never plays over a male
    voice. When ``voice`` is None we fall back to the env default,
    which keeps every other caller (chat mode, voice-only mode without
    avatar selection, dev/test scripts) working unchanged.
    """
    text = (text or "").strip()
    if not text:
        return

    fmt = (_s.openai_tts_format or "pcm").lower()
    chosen_voice = (voice or _s.openai_tts_voice or "alloy").strip()
    try:
        client = _async_client()
        async with client.audio.speech.with_streaming_response.create(
            model=_s.openai_tts_model,
            voice=chosen_voice,
            input=text,
            response_format=fmt,
            speed=1.0,
        ) as response:
            async for chunk in response.iter_bytes(chunk_size=4096):
                if chunk:
                    yield base64.b64encode(chunk).decode("ascii")
    except Exception as exc:  # noqa: BLE001
        log.error("[OpenAI] stream_tts_chunks failed (voice=%s): %s",
                  chosen_voice, exc)


# ─────────────────────────────────────────────────────────────────────────────
# Role Play — Whisper transcription
# ─────────────────────────────────────────────────────────────────────────────

async def transcribe_audio(raw_bytes: bytes, language_code: str = "en") -> str:
    """Transcribe raw audio bytes (webm/opus from MediaRecorder) via Whisper.

    ``language_code`` is the scenario's primary language. We treat it
    as a SOFT preference, not a hard filter — if the candidate speaks
    Hindi during an English-language scenario the transcript should
    still be in Hindi so the AI character can mirror the language.

    Implementation: we let Whisper auto-detect (no `language` param)
    so any language in the candidate's training catalogue works. The
    scenario language_code is only echoed in the log line for
    debuggability; it does NOT constrain transcription. This is what
    enables the "speak any language, AI replies in same language"
    multilingual flow described in the role-play spec.
    """
    if not raw_bytes:
        return ""
    try:
        client = _async_client()
        audio_file = BytesIO(raw_bytes)
        audio_file.name = "audio.webm"

        # Verbose JSON gives us the detected language back, which we
        # log so the operator can confirm Whisper is identifying e.g.
        # Hindi correctly. Falls back to text on older OpenAI client
        # versions that don't support verbose_json on the streaming
        # transcriptions endpoint.
        try:
            transcript = await client.audio.transcriptions.create(
                model=_s.openai_stt_model,
                file=audio_file,
                response_format="verbose_json",
            )
            text          = getattr(transcript, "text", "") or ""
            detected_lang = getattr(transcript, "language", "") or "?"
        except Exception:  # noqa: BLE001
            # Re-open the file handle (the previous call consumed it)
            # and fall back to plain text response_format.
            audio_file = BytesIO(raw_bytes)
            audio_file.name = "audio.webm"
            transcript = await client.audio.transcriptions.create(
                model=_s.openai_stt_model,
                file=audio_file,
                response_format="text",
            )
            text = (transcript or "")
            detected_lang = "?"
        text = (text or "").strip()
        log.info(
            "[STT] transcribed %d chars  detected_lang=%s  scenario_lang=%s",
            len(text), detected_lang, (language_code or "en"),
        )
        return text
    except Exception as exc:  # noqa: BLE001
        log.error("[OpenAI] transcribe_audio failed: %s", exc)
        return ""


# ─────────────────────────────────────────────────────────────────────────────
# Role Play — Live sentiment analysis
# ─────────────────────────────────────────────────────────────────────────────

_SENTIMENT_PROMPT = """Analyse this learner utterance from a professional training roleplay.
Return ONLY a valid JSON object — no markdown, no explanation:

{
  "score": <integer 0-100, overall quality of this response>,
  "tone": "<Professional|Empathetic|Calm|Neutral|Tense|Defensive|Aggressive>",
  "confidence": "<High|Medium|Low>",
  "structure": "<Excellent|Clear|Unclear>",
  "empathy": "<Strong|Moderate|Weak>",
  "pace": "<Fast|Normal|Slow>",
  "vocabulary": "<Professional|Good|Simple>",
  "keywords": ["<key phrase 1>", "<key phrase 2>"],
  "tips": [
    {"type": "<strength|coaching|focus|framework>", "label": "<short title>", "text": "<actionable tip>"}
  ]
}

Rules: Always return 2-3 tips. Use exactly one of each type per response.
Keep each tip specific to THIS utterance — never generic."""


async def analyse_sentiment(text: str) -> dict:
    """Analyse a learner utterance and return live coaching metrics."""
    if not (text or "").strip():
        return {}
    try:
        client = _async_client()
        resp = await client.chat.completions.create(
            model="gpt-4o-mini",
            messages=[
                {"role": "system", "content": _SENTIMENT_PROMPT},
                {"role": "user",   "content": text},
            ],
            temperature=0,
            max_tokens=450,
        )
        raw = (resp.choices[0].message.content or "").strip()
        raw = raw.removeprefix("```json").removeprefix("```").removesuffix("```").strip()
        return json.loads(raw)
    except Exception as exc:  # noqa: BLE001
        log.warning("[OpenAI] analyse_sentiment failed: %s", exc)
        return {}


# ─────────────────────────────────────────────────────────────────────────────
# Role Play — End-of-session report
# ─────────────────────────────────────────────────────────────────────────────

_REPORT_PROMPT = """\
You are a strict, professional candidate-interview evaluator. Analyse the
roleplay conversation and return ONLY a valid JSON object — no markdown, no
explanation.

CRITICAL SCORING PRINCIPLES — read carefully:
  • Score the LEARNER's contribution only (the candidate). Never credit the
    learner for things the AI character said. Never inflate scores out of
    politeness. Be realistic and evidence-based.
  • If the learner produced no substantive content (silent, one-word
    replies, off-topic noise, or fewer than 3 meaningful sentences across
    the whole session) the overall_score MUST be ≤ 2.0, every dimension
    score MUST be ≤ 25, and sentiment_timeline values MUST be ≤ 30.
  • If the learner only partially engaged (some replies but missed the
    scenario goals, didn't address the AI's concerns, or was vague), cap
    the overall_score at 5.0 and dimension scores at 55.
  • Reserve dimension scores ≥ 70 for clear, demonstrated competence in
    the transcript. Reserve overall_score ≥ 7.5 for strong candidates
    who actually solved or progressed the scenario.
  • sentiment_timeline reflects the LEARNER's emotional/communicative
    quality over time, not the AI's. Silence or disengagement is low
    sentiment (10–30), not neutral 50.

Required schema:
{
  "overall_score": <float 0.0–10.0>,
  "verdict": "<pass|needs_improvement>",
  "summary": "<2-3 sentence honest paragraph; if the candidate didn't
              engage, say so plainly>",
  "communication_style": "<Assertive|Empathetic|Analytical|Direct|Passive|Mixed|None>",
  "focus_areas": ["<area 1>", "<area 2>"],
  "dimensions": [
    {"name": "<dimension>", "score": <0-100>, "level": "<high|mid|low>",
     "detail": "<observation grounded in the transcript>",
     "tip": "<improvement tip>"}
  ],
  "feedback": [
    {"type": "<strength|weakness|suggestion>", "text": "<feedback>",
     "example": "<quote or empty>", "priority": "<high|medium|low>"}
  ],
  "key_moments": [
    {"turn": <int>, "description": "<what happened>",
     "impact": "<positive|negative|neutral>"}
  ],
  "sentiment_timeline": [<int 0-100>, ...],
  "freedom_practice_suggestions": ["<suggestion 1>", "<suggestion 2>"]
}

Dimensions to score (always include all 5):
  Communication Clarity, Empathy & Tone, Problem-Solving, Professionalism, Active Listening

verdict: "pass" if overall_score >= 6.5, else "needs_improvement".
freedom_practice_suggestions: 2-3 specific scenarios the learner should rehearse next.
"""


def _no_participation_report(learner_role: str, reason: str) -> dict:
    """Realistic report for sessions where the candidate barely engaged.

    Used as a hard guard so the LLM cannot return a falsely positive report
    when the transcript shows the learner never spoke. Mirrors the schema
    in `_REPORT_PROMPT` so downstream code (PDF render, dashboard) sees
    consistent shape.
    """
    flat_low = [
        {"name": "Communication Clarity", "score": 5,  "level": "low",
         "detail": "No spoken or written responses from the candidate were captured.",
         "tip": "Engage with the scenario — respond to the AI character to demonstrate clarity."},
        {"name": "Empathy & Tone", "score": 5,  "level": "low",
         "detail": "Cannot evaluate empathy without candidate input.",
         "tip": "Acknowledge the AI character's concern in your own words."},
        {"name": "Problem-Solving", "score": 0,  "level": "low",
         "detail": "No attempt was made to address the scenario's problem.",
         "tip": "Listen for the issue, then propose a concrete next step."},
        {"name": "Professionalism", "score": 10, "level": "low",
         "detail": "Disengagement during a candidate interview reads as unprofessional.",
         "tip": "Treat every prompt as a real interview question — respond promptly."},
        {"name": "Active Listening", "score": 5,  "level": "low",
         "detail": "No paraphrasing, clarifying, or follow-up questions were observed.",
         "tip": "Reflect back what the AI said before answering."},
    ]
    return {
        "overall_score": 1.0,
        "verdict": "needs_improvement",
        "summary": (
            f"The candidate did not engage with the scenario — {reason}. "
            f"There is no evidence of communication, problem-solving, or "
            f"interview readiness in this session. A realistic interview "
            f"outcome would be to not advance the candidate."
        ),
        "communication_style": "None",
        "focus_areas": ["Active participation", "Verbal response under pressure"],
        "dimensions": flat_low,
        "feedback": [
            {"type": "weakness", "text": "Candidate did not respond during the session.",
             "example": "", "priority": "high"},
            {"type": "suggestion",
             "text": (f"Re-attempt the scenario as the {learner_role} and respond "
                      f"to every AI prompt — even short replies are better than silence."),
             "example": "", "priority": "high"},
        ],
        "key_moments": [],
        # Low and flat — silence is not neutral, it's a negative signal.
        "sentiment_timeline": [15, 12, 10, 10, 8],
        "freedom_practice_suggestions": [
            "Cold-open practice: respond within 5 seconds of the AI's first prompt.",
            "Mirror-and-answer drill: paraphrase the AI's question, then answer it.",
        ],
    }


def _learner_engagement(messages, learner_role_label: str) -> tuple[int, int, list[str]]:
    """Return (count_of_meaningful_messages, total_words, raw_texts)."""
    meaningful = 0
    total_words = 0
    raw: list[str] = []
    for m in messages:
        role = getattr(m, "role", None)
        if role != "user":
            continue
        content = (getattr(m, "content", "") or "").strip()
        if not content or content.startswith("[") and content.endswith("]"):
            continue  # skip system markers like [session_start]
        raw.append(content)
        words = [w for w in re.split(r"\s+", content) if w]
        total_words += len(words)
        # "Meaningful" = at least 3 words OR ends a complete clause
        if len(words) >= 3:
            meaningful += 1
    return meaningful, total_words, raw


async def generate_report_analysis(messages, scenario, duration_seconds: int) -> dict:
    """Generate a full performance report from the session transcript.

    Realism guard:
      Before calling the LLM we count how much the candidate actually
      contributed. If they were silent or only produced trivial input
      (no meaningful messages, or fewer than ~5 total words) we
      short-circuit to `_no_participation_report()` — that returns a
      transcript-faithful "candidate did not engage" report instead of
      letting the LLM hallucinate praise from an empty conversation.
    """
    g = (lambda k, default="": (
        getattr(scenario, k, None) if not isinstance(scenario, dict) else scenario.get(k)
    ) or default)

    learner_role = g("learner_role", "Candidate")
    ai_character = g("ai_character", "Character")

    # ── Hard realism guard ──────────────────────────────────────────────
    meaningful, total_words, _raw = _learner_engagement(messages, learner_role)
    if meaningful == 0 and total_words < 3:
        log.info(
            "[Report] no-participation guard fired (meaningful=%d words=%d) — "
            "returning realistic low-score report",
            meaningful, total_words,
        )
        return _no_participation_report(
            learner_role,
            reason="the candidate did not respond during the session",
        )

    lines = []
    for m in messages:
        content = getattr(m, "content", "") or ""
        if content in ("[session_start]",):
            continue
        speaker = learner_role if getattr(m, "role", None) == "user" else ai_character
        lines.append(f"{speaker}: {content}")
    transcript_text = "\n".join(lines) or "(empty session)"

    # Pre-compute participation hints so the LLM cannot ignore them.
    participation_note = (
        f"Candidate produced {meaningful} meaningful message(s) and "
        f"{total_words} total word(s). "
    )
    if meaningful <= 2 or total_words < 30:
        participation_note += (
            "This is LOW engagement — apply the strict scoring caps in the "
            "system prompt. Do not invent strengths that aren't in the transcript."
        )

    context = (
        f"Scenario: {g('title', 'Role Play')}\n"
        f"Learner role: {learner_role}\n"
        f"AI character: {ai_character}\n"
        f"Difficulty: {g('difficulty', 'Intermediate')}\n"
        f"Duration: {duration_seconds}s\n"
        f"{participation_note}\n\n"
        f"TRANSCRIPT:\n{transcript_text}"
    )

    # Brief, structured log of what we're sending OpenAI for sentiment
    # analysis. Intentionally NOT dumping the full transcript or
    # _REPORT_PROMPT here (those are large) — only the metadata that
    # answers "did the right turns / scenario reach the model".
    log.info(
        "[Report] OpenAI sentiment call → model=%s  scenario=%s  "
        "learner_role=%s  duration=%ds  meaningful=%d  total_words=%d",
        _s.openai_chat_model,
        g("title", default="(no title)"),
        learner_role,
        duration_seconds,
        meaningful,
        total_words,
    )
    try:
        client = _async_client()
        resp = await client.chat.completions.create(
            model=_s.openai_chat_model,
            messages=[
                {"role": "system", "content": _REPORT_PROMPT},
                {"role": "user",   "content": context},
            ],
            temperature=0.3,
            max_tokens=2000,
        )
        raw = (resp.choices[0].message.content or "").strip()
        raw = raw.removeprefix("```json").removeprefix("```").removesuffix("```").strip()
        data = json.loads(raw)
        # Headline result — score + verdict + the per-dimension names
        # so the operator can quickly verify the LLM returned a
        # well-formed report without dumping the entire JSON.
        log.info(
            "[Report] OpenAI sentiment result ← score=%.1f  verdict=%s  "
            "turns=%d  dimensions=%s  feedback_count=%d",
            float(data.get("overall_score", 0) or 0),
            data.get("verdict", "?"),
            len([m for m in messages if getattr(m, "role", None) == "user"]),
            [d.get("name") for d in (data.get("dimensions") or [])] or "(none)",
            len(data.get("feedback") or []),
        )
        return data
    except Exception as exc:  # noqa: BLE001
        log.error("[OpenAI] generate_report_analysis failed: %s", exc)
        return {
            "overall_score": 5.0,
            "verdict": "needs_improvement",
            "summary": "Report generation failed — fallback values shown.",
            "dimensions": [
                {"name": "Communication Clarity", "score": 50, "level": "mid", "detail": "", "tip": ""},
                {"name": "Empathy & Tone",        "score": 50, "level": "mid", "detail": "", "tip": ""},
                {"name": "Problem-Solving",       "score": 50, "level": "mid", "detail": "", "tip": ""},
                {"name": "Professionalism",       "score": 50, "level": "mid", "detail": "", "tip": ""},
                {"name": "Active Listening",      "score": 50, "level": "mid", "detail": "", "tip": ""},
            ],
            "feedback": [
                {"type": "suggestion", "text": f"Report generation failed: {exc}. Please retry.",
                 "example": "", "priority": "high"},
            ],
            "key_moments": [],
            "sentiment_timeline": [50],
            "freedom_practice_suggestions": [],
        }


# ─────────────────────────────────────────────────────────────────────────────
# AI Scenario generator (admin)
# ─────────────────────────────────────────────────────────────────────────────

_SCENARIO_GEN_PROMPT = """\
You are a learning-design expert for a corporate training platform.
Given a description, return ONLY a valid JSON object for one roleplay scenario.

Required schema (all fields required):
{
  "title": "<max 60 chars>",
  "description": "<2-3 sentences>",
  "context": "<3-5 sentences>",
  "category": "<cs|hr|sales|mgmt|comp|finance>",
  "cat_label": "<Human-readable>",
  "learner_role": "<job title>",
  "learner_emoji": "<single emoji>",
  "ai_character": "<name + role>",
  "ai_emoji": "<single emoji>",
  "ai_personality": "<2-3 sentences>",
  "difficulty": "<Beginner|Intermediate|Advanced>",
  "turns": <int 6-12>,
  "time_limit": "<e.g. 10 min>",
  "scoring": "Guided",
  "voice_on": true,
  "language_code": "en",
  "language_name": "English",
  "language_flag": "🇬🇧",
  "featured": false
}

Return JSON only — no markdown, no explanation."""


async def generate_scenario_from_prompt(prompt: str) -> dict:
    """Generate a complete scenario definition from a free-text admin prompt."""
    try:
        client = _async_client()
        resp = await client.chat.completions.create(
            model=_s.openai_chat_model,
            messages=[
                {"role": "system", "content": _SCENARIO_GEN_PROMPT},
                {"role": "user",   "content": prompt},
            ],
            temperature=0.7,
            max_tokens=700,
        )
        raw = (resp.choices[0].message.content or "").strip()
        raw = raw.removeprefix("```json").removeprefix("```").removesuffix("```").strip()
        return json.loads(raw)
    except Exception as exc:  # noqa: BLE001
        log.error("[OpenAI] generate_scenario_from_prompt failed: %s", exc)
        raise RuntimeError(f"AI scenario generation failed: {exc}") from exc
