import os
import requests
import json
import logging
import tempfile
import librosa
import numpy as np
import base64
import speech_recognition as sr
from pydub import AudioSegment
from typing import Dict, Any, Optional

# ---------------------------------------------------------------------------
# Logging
# ---------------------------------------------------------------------------
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger("response_iq_analysis")

# ---------------------------------------------------------------------------
# Ollama Configuration
# ---------------------------------------------------------------------------
OLLAMA_API_URL = "http://localhost:11434/api/generate"
OLLAMA_MODEL   = "qwen2.5:7b"   # Superior multilingual support for Indian scripts

# ---------------------------------------------------------------------------
# Speech Recognition Configuration
# ---------------------------------------------------------------------------
_recognizer = sr.Recognizer()

# ---------------------------------------------------------------------------
# Language Configuration Registry
#
# Each language entry defines:
#   display_name         – Human-readable name shown in logs and API docs.
#   script_name          – The script/writing system name (used in prompts).
#   response_instruction – Hard instruction appended to system & user prompt
#                          telling the LLM which language/script to use.
#   audio_context_header – Heading for the voice-metadata block in the prompt.
#   audio_hint           – Native-language explanation of acoustic stats.
#   text_tone_fallback   – Instruction when no voice is available.
#   visuals_instruction  – How the 3 vibe-keywords should be written.
# ---------------------------------------------------------------------------
LANGUAGE_CONFIG: Dict[str, Dict[str, str]] = {

    "english": {
        "display_name": "English",
        "script_name": "English (Latin script)",
        "speech_locale": "en-IN",
        "response_instruction": (
            "Write ALL analysis fields — comparison, behavioralAnalysis, tonalAnalysis — "
            "entirely in English. The 'visuals' keywords must also be in English."
        ),
        "audio_context_header": "Voice Tonal Metadata:",
        "audio_hint": (
            "Stat interpretation guide — "
            "Low pitch_std = monotone delivery; "
            "High energy_fluctuation = shaky/hesitant voice; "
            "High avg_flatness = breathy or unclear articulation; "
            "Moderate tempo = confident and composed delivery."
        ),
        "text_tone_fallback": (
            "Since no voice metadata is available, base the tonal analysis on the "
            "written tone of the answer (e.g., formal, casual, assertive, apologetic)."
        ),
        "visuals_instruction": "3 single English words or short English phrases.",
    },

    "hindi": {
        "display_name": "Hindi",
        "script_name": "Hindi (Devanagari script / हिन्दी)",
        "speech_locale": "hi-IN",
        "response_instruction": (
            "The question, ideal answer, and user's answer are all in Hindi (Devanagari script). "
            "You MUST write ALL analysis fields — comparison, behavioralAnalysis, tonalAnalysis — "
            "entirely in Hindi using Devanagari script. "
            "Do NOT mix English sentences inside the analysis fields. "
            "Technical terms may be kept in English only when no Hindi equivalent exists."
        ),
        "audio_context_header": "आवाज़ की टोनल मेटाडेटा (Voice Tonal Metadata):",
        "audio_hint": (
            "स्टैट्स की व्याख्या — "
            "कम pitch_std = एकरस/नीरस आवाज़; "
            "अधिक energy_fluctuation = कंपकंपाती या हिचकिचाती आवाज़; "
            "अधिक avg_flatness = सांसयुक्त या अस्पष्ट उच्चारण; "
            "मध्यम tempo = आत्मविश्वासपूर्ण और संयमित बोलने का तरीका।"
        ),
        "text_tone_fallback": (
            "चूँकि कोई आवाज़ मेटाडेटा उपलब्ध नहीं है, टोनल विश्लेषण उत्तर की "
            "लिखित भाषा-शैली के आधार पर करें (जैसे — औपचारिक, अनौपचारिक, दृढ़, विनम्र)।"
        ),
        "visuals_instruction": (
            "3 हिंदी शब्द या छोटे हिंदी वाक्यांश जो उत्तर का समग्र भाव दर्शाएं।"
        ),
    },

    "marathi": {
        "display_name": "Marathi",
        "script_name": "Marathi (Devanagari script / मराठी)",
        "speech_locale": "mr-IN",
        "response_instruction": (
            "The question, ideal answer, and user's answer are all in Marathi (Devanagari script). "
            "You MUST write ALL analysis fields — comparison, behavioralAnalysis, tonalAnalysis — "
            "entirely in Marathi using Devanagari script. "
            "Do NOT mix English sentences inside the analysis fields. "
            "Technical terms may be kept in English only when no Marathi equivalent exists."
        ),
        "audio_context_header": "आवाजाचे टोनल मेटाडेटा (Voice Tonal Metadata):",
        "audio_hint": (
            "स्टॅट्सचा अर्थ — "
            "कमी pitch_std = एकसुरी/नीरस आवाज; "
            "जास्त energy_fluctuation = थरथरणारा किंवा अडखळणारा आवाज; "
            "जास्त avg_flatness = श्वासयुक्त किंवा अस्पष्ट उच्चार; "
            "मध्यम tempo = आत्मविश्वासपूर्ण आणि संयमित बोलण्याची पद्धत।"
        ),
        "text_tone_fallback": (
            "कारण कोणताही आवाज मेटाडेटा उपलब्ध नाही, टोनल विश्लेषण उत्तराच्या "
            "लिखित शैलीवर आधारित करा (उदा. औपचारिक, अनौपचारिक, ठाम, नम्र)."
        ),
        "visuals_instruction": (
            "3 मराठी शब्द किंवा छोटे मराठी वाक्यांश जे उत्तराचा एकूण भाव दर्शवतात."
        ),
    },

    "tamil": {
        "display_name": "Tamil",
        "script_name": "Tamil (Tamil script / தமிழ்)",
        "speech_locale": "ta-IN",
        "response_instruction": (
            "The question, ideal answer, and user's answer are all in Tamil (Tamil script). "
            "You MUST write ALL analysis fields — comparison, behavioralAnalysis, tonalAnalysis — "
            "entirely in Tamil using Tamil script. "
            "Do NOT mix English sentences inside the analysis fields. "
            "Technical terms may be kept in English only when no Tamil equivalent exists."
        ),
        "audio_context_header": "குரல் டோனல் மெட்டாடேட்டா (Voice Tonal Metadata):",
        "audio_hint": (
            "புள்ளிவிவர விளக்கம் — "
            "குறைந்த pitch_std = ஒரே சுரத்தில் பேசும் முறை; "
            "அதிக energy_fluctuation = நடுங்கும் அல்லது தயங்கும் குரல்; "
            "அதிக avg_flatness = மூச்சுக்காற்று கலந்த அல்லது தெளிவற்ற உச்சரிப்பு; "
            "மிதமான tempo = தன்னம்பிக்கையான மற்றும் அமைதியான பேச்சு முறை."
        ),
        "text_tone_fallback": (
            "குரல் மெட்டாடேட்டா இல்லாததால், டோனல் பகுப்பாய்வை "
            "எழுதப்பட்ட பதிலின் தொனியை அடிப்படையாகக் கொண்டு செய்யவும் "
            "(எ.கா. அலுவல்முறை, சாதாரண, உறுதியான, பணிவான)."
        ),
        "visuals_instruction": (
            "3 தமிழ் வார்த்தைகள் அல்லது சிறிய தமிழ் சொற்றொடர்கள் "
            "பதிலின் ஒட்டுமொத்த தன்மையை விவரிக்கும்."
        ),
    },

    "telugu": {
        "display_name": "Telugu",
        "script_name": "Telugu (Telugu script / తెలుగు)",
        "speech_locale": "te-IN",
        "response_instruction": (
            "The question, ideal answer, and user's answer are all in Telugu (Telugu script). "
            "You MUST write ALL analysis fields — comparison, behavioralAnalysis, tonalAnalysis — "
            "entirely in Telugu using Telugu script. "
            "Do NOT mix English sentences inside the analysis fields. "
            "Technical terms may be kept in English only when no Telugu equivalent exists."
        ),
        "audio_context_header": "వాయిస్ టోనల్ మెటాడేటా (Voice Tonal Metadata):",
        "audio_hint": (
            "గణాంక వివరణ — "
            "తక్కువ pitch_std = ఏకస్వర పద్ధతిలో మాట్లాడటం; "
            "అధిక energy_fluctuation = వణికే లేదా సంకోచించే స్వరం; "
            "అధిక avg_flatness = శ్వాసతో కూడిన లేదా అస్పష్టమైన ఉచ్చారణ; "
            "మధ్యస్థ tempo = ఆత్మవిశ్వాసంతో మరియు స్థిరంగా మాట్లాడడం."
        ),
        "text_tone_fallback": (
            "వాయిస్ మెటాడేటా అందుబాటులో లేనందున, టోనల్ విశ్లేషణను "
            "రాసిన సమాధానం యొక్క శైలి ఆధారంగా చేయండి "
            "(ఉదా. అధికారిక, అనధికారిక, నిర్ణయాత్మక, వినయంగా)."
        ),
        "visuals_instruction": (
            "3 తెలుగు పదాలు లేదా చిన్న తెలుగు పదబంధాలు "
            "సమాధానం యొక్క మొత్తం భావాన్ని వ్యక్తం చేసేవి."
        ),
    },

    "kannada": {
        "display_name": "Kannada",
        "script_name": "Kannada (Kannada script / ಕನ್ನಡ)",
        "speech_locale": "kn-IN",
        "response_instruction": (
            "The question, ideal answer, and user's answer are all in Kannada (Kannada script). "
            "You MUST write ALL analysis fields — comparison, behavioralAnalysis, tonalAnalysis — "
            "entirely in Kannada using Kannada script. "
            "Do NOT mix English sentences inside the analysis fields. "
            "Technical terms may be kept in English only when no Kannada equivalent exists."
        ),
        "audio_context_header": "ಧ್ವನಿ ಟೋನಲ್ ಮೆಟಾಡೇಟಾ (Voice Tonal Metadata):",
        "audio_hint": (
            "ಅಂಕಿ-ಅಂಶಗಳ ವಿವರಣೆ — "
            "ಕಡಿಮೆ pitch_std = ಏಕತಾನ ಮಾತಿನ ರೀತಿ; "
            "ಹೆಚ್ಚು energy_fluctuation = ನಡುಗುವ ಅಥವಾ ಅಳುಕುವ ಧ್ವನಿ; "
            "ಹೆಚ್ಚು avg_flatness = ಉಸಿರು ಸೇರಿದ ಅಥವಾ ಅಸ್ಪಷ್ಟ ಉಚ್ಚಾರಣೆ; "
            "ಮಧ್ಯಮ tempo = ಆತ್ಮವಿಶ್ವಾಸದಿಂದ ಮತ್ತು ಸ್ಥಿರವಾಗಿ ಮಾತನಾಡುವ ರೀತಿ."
        ),
        "text_tone_fallback": (
            "ಧ್ವನಿ ಮೆಟಾಡೇಟಾ ಲಭ್ಯವಿಲ್ಲದ ಕಾರಣ, ಟೋನಲ್ ವಿಶ್ಲೇಷಣೆಯನ್ನು "
            "ಬರೆದ ಉತ್ತರದ ಶೈಲಿಯ ಆಧಾರದ ಮೇಲೆ ಮಾಡಿ "
            "(ಉದಾ. ಔಪಚಾರಿಕ, ಅನೌಪಚಾರಿಕ, ದೃಢ, ವಿನಮ್ರ)."
        ),
        "visuals_instruction": (
            "3 ಕನ್ನಡ ಪದಗಳು ಅಥವಾ ಚಿಕ್ಕ ಕನ್ನಡ ಪದಗುಚ್ಛಗಳು "
            "ಉತ್ತರದ ಒಟ್ಟಾರೆ ಭಾವವನ್ನು ಅಭಿವ್ಯಕ್ತಿಸುವ."
        ),
    },

    "punjabi": {
        "display_name": "Punjabi",
        "script_name": "Punjabi (Gurmukhi script / ਪੰਜਾਬੀ)",
        "speech_locale": "pa-IN",
        "response_instruction": (
            "The question, ideal answer, and user's answer are all in Punjabi (Gurmukhi script). "
            "You MUST write ALL analysis fields — comparison, behavioralAnalysis, tonalAnalysis — "
            "entirely in Punjabi using Gurmukhi script. "
            "Do NOT mix English sentences inside the analysis fields. "
            "Technical terms may be kept in English only when no Punjabi equivalent exists."
        ),
        "audio_context_header": "ਆਵਾਜ਼ ਦਾ ਟੋਨਲ ਮੈਟਾਡੇਟਾ (Voice Tonal Metadata):",
        "audio_hint": (
            "ਅੰਕੜਿਆਂ ਦੀ ਵਿਆਖਿਆ — "
            "ਘੱਟ pitch_std = ਇੱਕਸੁਰੀ ਬੋਲਣ ਦਾ ਢੰਗ; "
            "ਵੱਧ energy_fluctuation = ਕੰਬਦੀ ਜਾਂ ਝਿਜਕਦੀ ਆਵਾਜ਼; "
            "ਵੱਧ avg_flatness = ਸਾਹ ਭਰੀ ਜਾਂ ਅਸਪੱਸ਼ਟ ਉਚਾਰਨ; "
            "ਦਰਮਿਆਨੀ tempo = ਆਤਮਵਿਸ਼ਵਾਸ ਨਾਲ ਅਤੇ ਸੰਜਮ ਨਾਲ ਬੋਲਣਾ।"
        ),
        "text_tone_fallback": (
            "ਕਿਉਂਕਿ ਕੋਈ ਆਵਾਜ਼ ਮੈਟਾਡੇਟਾ ਉਪਲਬਧ ਨਹੀਂ ਹੈ, ਟੋਨਲ ਵਿਸ਼ਲੇਸ਼ਣ "
            "ਲਿਖਤੀ ਜਵਾਬ ਦੀ ਭਾਸ਼ਾ-ਸ਼ੈਲੀ ਦੇ ਆਧਾਰ 'ਤੇ ਕਰੋ "
            "(ਜਿਵੇਂ — ਰਸਮੀ, ਗੈਰ-ਰਸਮੀ, ਦ੍ਰਿੜ੍ਹ, ਨਿਮਰ)।"
        ),
        "visuals_instruction": (
            "3 ਪੰਜਾਬੀ ਸ਼ਬਦ ਜਾਂ ਛੋਟੇ ਪੰਜਾਬੀ ਵਾਕਾਂਸ਼ ਜੋ ਜਵਾਬ ਦੇ ਸਮੁੱਚੇ ਭਾਵ ਨੂੰ ਦਰਸਾਉਂਦੇ ਹੋਣ।"
        ),
    },

    "hinglish": {
        "display_name": "Hinglish",
        "script_name": "Hinglish (mix of Hindi and English in Latin script)",
        "speech_locale": "hi-IN",
        "response_instruction": (
            "The question, ideal answer, and user's answer are in Hinglish (a mix of Hindi and English written in Latin script). "
            "You MUST write ALL analysis fields — comparison, behavioralAnalysis, tonalAnalysis — "
            "entirely in Hinglish using Latin script. "
            "Do NOT write in Devanagari script. Use conversational, professional Hinglish."
        ),
        "audio_context_header": "Voice Tonal Metadata (आवाज़ की टोनल जानकारी):",
        "audio_hint": (
            "Stats interpretation — "
            "कम pitch_std = monotone/एकरस delivery; "
            "high energy_fluctuation = shaky/hesitant voice; "
            "high avg_flatness = breathy/अस्पष्ट articulation; "
            "moderate tempo = confident/संयमित delivery."
        ),
        "text_tone_fallback": (
            "Since no voice metadata is available, base the tonal analysis on the "
            "written tone of the answer (e.g. formal, casual, confident, polite) but write the analysis in Hinglish."
        ),
        "visuals_instruction": (
            "3 Hinglish words or short Hinglish phrases capturing the overall vibe."
        ),
    },

    "bengali": {
        "display_name": "Bengali",
        "script_name": "Bengali (Bengali script / বাংলা)",
        "speech_locale": "bn-IN",
        "response_instruction": (
            "The question, ideal answer, and user's answer are all in Bengali (Bengali script). "
            "You MUST write ALL analysis fields — comparison, behavioralAnalysis, tonalAnalysis — "
            "entirely in Bengali using Bengali script. "
            "Do NOT mix English sentences inside the analysis fields. "
            "Technical terms may be kept in English only when no Bengali equivalent exists."
        ),
        "audio_context_header": "ভয়েস টোনাল মেটাডেটা (Voice Tonal Metadata):",
        "audio_hint": (
            "পরিসংখ্যানের ব্যাখ্যা — "
            "কম pitch_std = একঘেয়ে কথা বলার ধরন; "
            "বেশি energy_fluctuation = কাঁপানো বা দ্বিধাগ্রস্ত কণ্ঠস্বর; "
            "বেশি avg_flatness = শ্বাসযুক্ত বা অস্পষ্ট উচ্চারণ; "
            "মাঝারি tempo = আত্মবিশ্বাসী এবং সংযত কথা বলার ধরন।"
        ),
        "text_tone_fallback": (
            "যেহেতু কোনো ভয়েস মেটাডেটা উপলব্ধ নেই, তাই টোনাল বিশ্লেষণ উত্তরের "
            "লিখিত শৈলীর ওপর ভিত্তি করে করুন (যেমন আনুষ্ঠানিক, অনানুষ্ঠানিক, দৃঢ়, নম্র)।"
        ),
        "visuals_instruction": (
            "৩টি বাংলা শব্দ বা ছোট বাংলা বাক্যাংশ যা উত্তরের সামগ্রিক ভাব প্রকাশ করে।"
        ),
    },

    "gujarati": {
        "display_name": "Gujarati",
        "script_name": "Gujarati (Gujarati script / ગુજરાતી)",
        "speech_locale": "gu-IN",
        "response_instruction": (
            "The question, ideal answer, and user's answer are all in Gujarati (Gujarati script). "
            "You MUST write ALL analysis fields — comparison, behavioralAnalysis, tonalAnalysis — "
            "entirely in Gujarati using Gujarati script. "
            "Do NOT mix English sentences inside the analysis fields. "
            "Technical terms may be kept in English only when no Gujarati equivalent exists."
        ),
        "audio_context_header": "અવાજની ટોનલ મેટાડેટા (Voice Tonal Metadata):",
        "audio_hint": (
            "આંકડાકીય સમજૂતી — "
            "ઓછું pitch_std = એકરસ બોલવાની રીત; "
            "વધુ energy_fluctuation = ધ્રૂજતો કે અચકાતો અવાજ; "
            "વધુ avg_flatness = શ્વાસવાળો કે અસ્પષ્ટ ઉચ્ચાર; "
            "મધ્યમ tempo = આત્મવિશ્વાસપૂર્ણ અને સંયમિત બોલવાની રીત."
        ),
        "text_tone_fallback": (
            "કોઈ અવાજની મેટાડેટા ઉપલબ્ધ ન હોવાથી, ટોનલ વિશ્લેષણ જવાબની "
            "લેખિત શૈલીના આધારે કરો (દા.ત. ઔપચારિક, અનૌપચારિક, દૃઢ, નમ્ર)."
        ),
        "visuals_instruction": (
            "૩ ગુજરાતી શબ્દો અથવા નાના ગુજરાતી શબ્દસમૂહો જે જવાબનો એકંદર ભાવ દર્શાવે છે."
        ),
    },

    "malayalam": {
        "display_name": "Malayalam",
        "script_name": "Malayalam (Malayalam script / മലയാളം)",
        "speech_locale": "ml-IN",
        "response_instruction": (
            "The question, ideal answer, and user's answer are all in Malayalam (Malayalam script). "
            "You MUST write ALL analysis fields — comparison, behavioralAnalysis, tonalAnalysis — "
            "entirely in Malayalam using Malayalam script. "
            "Do NOT mix English sentences inside the analysis fields. "
            "Technical terms may be kept in English only when no Malayalam equivalent exists."
        ),
        "audio_context_header": "ശബ്ദ ടോണൽ മെറ്റാഡാറ്റ (Voice Tonal Metadata):",
        "audio_hint": (
            "സ്ഥിതിവിവരക്കണക്ക് വിശദീകരണം — "
            "കുറഞ്ഞ pitch_std = ഒരേ ഈണത്തിലുള്ള സംസാരം; "
            "കൂടിയ energy_fluctuation = വിറയ്ക്കുന്ന അല്ലെങ്കിൽ ഇടറുന്ന ശബ്ദം; "
            "കൂടിയ avg_flatness = വ്യക്തമല്ലാത്ത ഉച്ചാരണം; "
            "മിതമായ tempo = തികഞ്ഞ ആത്മവിശ്വാസത്തോടെയുള്ള സംസാരം."
        ),
        "text_tone_fallback": (
            "ശബ്ദ മെറ്റാഡാറ്റ ലഭ്യമല്ലാത്തതിനാൽ, ടോണൽ വിശകലനം "
            "എഴുതിയ ഉത്തരത്തിന്റെ ശൈലിയെ അടിസ്ഥാനമാക്കി ചെയ്യുക "
            "(ഉദാ: ഔദ്യോഗികം, സൗഹൃദപരം, ഉറച്ച, വിനയപൂർവ്വം)."
        ),
        "visuals_instruction": (
            "ഉത്തരത്തിന്റെ മൊത്തത്തിലുള്ള ഭാവം വ്യക്തമാക്കുന്ന 3 മലയാളം വാക്കുകൾ അല്ലെങ്കിൽ ചെറിയ വാക്യങ്ങൾ."
        ),
    },

    "odia": {
        "display_name": "Odia",
        "script_name": "Odia (Odia script / ଓଡ଼ିଆ)",
        "speech_locale": "or-IN",
        "response_instruction": (
            "The question, ideal answer, and user's answer are all in Odia (Odia script). "
            "You MUST write ALL analysis fields — comparison, behavioralAnalysis, tonalAnalysis — "
            "entirely in Odia using Odia script. "
            "Do NOT mix English sentences inside the analysis fields. "
            "Technical terms may be kept in English only when no Odia equivalent exists."
        ),
        "audio_context_header": "ସ୍ୱର ଟୋନାଲ୍ ମେଟାଡାଟା (Voice Tonal Metadata):",
        "audio_hint": (
            "ପରିସଂଖ୍ୟାନର ବ୍ୟାଖ୍ୟା — "
            "କମ୍ pitch_std = ଏକସ୍ୱର କଥାବାର୍ତ୍ତା; "
            "ଅଧିକ energy_fluctuation = ଥରୁଥିବା କିମ୍ବା କୁଣ୍ଠିତ ସ୍ୱର; "
            "ଅଧିକ avg_flatness = ଅସ୍ପଷ୍ଟ ଉଚ୍ଚାରଣ; "
            "ମଧ୍ୟମ tempo = ଆତ୍ମବିଶ୍ୱାସପୂର୍ଣ୍ଣ ଏବଂ ସଂଯମ କଥାବାର୍ତ୍ତା।"
        ),
        "text_tone_fallback": (
            "ଯେହେତୁ କୌଣସି ସ୍ୱର ମେଟାଡାଟା ଉପଲବ୍ଧ ନାହିଁ, ଟୋନାଲ୍ ବିଶ୍ଳେଷଣ ଉତ୍ତରର "
            "ଲିଖିତ ଶୈଳୀ ଉପରେ ଆଧାର କରି କରନ୍ତୁ "
            "(ଉଦାହରଣ ସ୍ୱରୂପ: ଆନୁଷ୍ଠାନିକ, ଅନୌପଚାରିକ, ଦୃଢ଼, ନମ୍ର)।"
        ),
        "visuals_instruction": (
            "୩ଟି ଓଡ଼ିଆ ଶବ୍ଦ କିମ୍ବା ଛୋଟ ଓଡ଼ିଆ ବାକ୍ୟାଂଶ ଯାହା ଉତ୍ତରର ସାମଗ୍ରିକ ଭାବ ପ୍ରକାଶ କରେ।"
        ),
    }
}

DEFAULT_LANGUAGE = "english"
SUPPORTED_LANGUAGES = list(LANGUAGE_CONFIG.keys())


# ---------------------------------------------------------------------------
# Helpers
# ---------------------------------------------------------------------------

def get_language_config(language: Optional[str]) -> Dict[str, str]:
    """
    Returns the language config dict for the given language key.
    Case-insensitive. Falls back gracefully to English with a warning log.
    """
    key = (language or DEFAULT_LANGUAGE).strip().lower()
    if key not in LANGUAGE_CONFIG:
        logger.warning(
            f"Unsupported language '{key}'. "
            f"Supported: {SUPPORTED_LANGUAGES}. Falling back to '{DEFAULT_LANGUAGE}'."
        )
        key = DEFAULT_LANGUAGE
    return LANGUAGE_CONFIG[key]


def download_or_save_audio(voice_input: str) -> str:
    """
    Accepts either a remote URL or a Base64-encoded audio string.
    Downloads / decodes it and saves to a temp file.
    Returns the temp file path, or empty string on failure.
    """
    try:
        if not voice_input:
            return ""

        # ── URL ──────────────────────────────────────────────────────────────
        if voice_input.startswith("http://") or voice_input.startswith("https://"):
            response = requests.get(voice_input, stream=True, timeout=60)
            response.raise_for_status()

            # Infer file extension from URL; default to .mpeg
            suffix = ".mpeg"
            url_path = voice_input.split("?")[0]
            last_segment = url_path.split("/")[-1]
            if "." in last_segment:
                suffix = "." + last_segment.split(".")[-1]

            with tempfile.NamedTemporaryFile(delete=False, suffix=suffix) as tmp:
                for chunk in response.iter_content(chunk_size=8192):
                    tmp.write(chunk)
                return tmp.name

        # ── Base64 ───────────────────────────────────────────────────────────
        else:
            if "," in voice_input:
                voice_input = voice_input.split(",")[1]

            audio_data = base64.b64decode(voice_input)
            with tempfile.NamedTemporaryFile(delete=False, suffix=".wav") as tmp:
                tmp.write(audio_data)
                return tmp.name

    except Exception as e:
        logger.error(f"Failed to process audio input: {e}")
        return ""


def _convert_to_wav(file_path: str) -> str:
    """
    Converts any audio file to WAV format (16-bit PCM, mono, 16kHz)
    required by speech_recognition. Uses pydub for format conversion.

    Returns the path to the converted WAV file.
    """
    try:
        wav_path = file_path.rsplit(".", 1)[0] + "_converted.wav"
        audio = AudioSegment.from_file(file_path)
        # Convert to mono, 16kHz, 16-bit PCM — optimal for speech recognition
        audio = audio.set_channels(1).set_frame_rate(16000).set_sample_width(2)
        audio.export(wav_path, format="wav")
        logger.info(f"Audio converted to WAV: {wav_path}")
        return wav_path
    except Exception as e:
        logger.error(f"Audio conversion failed: {e}")
        return file_path  # Fall back to original file


def transcribe_audio(file_path: str, language: Optional[str] = None) -> Dict[str, Any]:
    """
    Transcribes audio to text using Google Speech Recognition API
    via the speech_recognition library.

    Parameters
    ----------
    file_path : str
        Path to the audio file.
    language  : str, optional
        Language key (e.g. 'hindi', 'tamil'). Maps to Google's
        locale codes (hi-IN, ta-IN, etc.) for accurate recognition.

    Returns
    -------
    dict:
        'transcription'         – The transcribed text.
        'detected_language'     – Language locale used for recognition.
        'speaking_duration_sec' – Total audio duration in seconds.
        'words_per_minute'      – Estimated speaking rate.
    """
    wav_path = None
    try:
        # Get the language locale for Google Speech Recognition
        lang_cfg = get_language_config(language)
        speech_locale = lang_cfg.get("speech_locale", "en-IN")

        logger.info(
            f"Transcribing audio with Google Speech Recognition | "
            f"locale={speech_locale} | file={file_path}"
        )

        # Convert audio to WAV format for speech_recognition compatibility
        wav_path = _convert_to_wav(file_path)

        # Load audio and get duration using librosa
        y, sr_rate = librosa.load(wav_path, sr=None)
        total_duration = float(len(y)) / sr_rate

        # Use speech_recognition to transcribe
        recognizer = _recognizer
        with sr.AudioFile(wav_path) as source:
            # Adjust for ambient noise for better accuracy
            recognizer.adjust_for_ambient_noise(source, duration=0.5)
            audio_data = recognizer.record(source)

        # Try Google Speech Recognition (free, supports Indian languages)
        transcription_text = ""
        try:
            transcription_text = recognizer.recognize_google(
                audio_data,
                language=speech_locale,
                show_all=False,
            )
            if isinstance(transcription_text, dict):
                # If show_all was True, extract best alternative
                alternatives = transcription_text.get("alternative", [{}])
                transcription_text = alternatives[0].get("transcript", "") if alternatives else ""
            transcription_text = str(transcription_text).strip()
        except sr.UnknownValueError:
            logger.warning("Google Speech Recognition could not understand the audio.")
            transcription_text = ""
        except sr.RequestError as e:
            logger.error(f"Google Speech Recognition API error: {e}")
            transcription_text = ""

        # Compute speaking metrics
        word_count = len(transcription_text.split()) if transcription_text else 0
        words_per_minute = round((word_count / max(total_duration, 1)) * 60, 1)

        logger.info(
            f"Transcription complete | "
            f"locale={speech_locale} | "
            f"text_length={len(transcription_text)} | "
            f"duration={total_duration:.1f}s | "
            f"wpm={words_per_minute}"
        )

        return {
            "transcription": transcription_text,
            "detected_language": speech_locale,
            "speaking_duration_sec": round(total_duration, 2),
            "words_per_minute": words_per_minute,
        }

    except Exception as e:
        logger.error(f"Speech transcription failed: {e}")
        return {
            "transcription": "",
            "detected_language": "unknown",
            "speaking_duration_sec": 0.0,
            "words_per_minute": 0.0,
            "error": str(e),
        }
    finally:
        # Clean up converted WAV file
        if wav_path and wav_path != file_path:
            try:
                os.remove(wav_path)
            except Exception:
                pass

def analyze_audio_features(file_path: str) -> Dict[str, Any]:
    """
    Extracts acoustic features using librosa.

    Returns
    -------
    dict:
        'summary' – Objective English description (LLM re-interprets in target language).
        'stats'   – Raw numeric features (language-agnostic).
    """
    try:
        y, sr = librosa.load(file_path, sr=None)

        # ── 1. Pitch ─────────────────────────────────────────────────────────
        pitches, magnitudes = librosa.piptrack(y=y, sr=sr)
        pitches_masked = pitches[magnitudes > np.median(magnitudes)]
        avg_pitch = float(np.mean(pitches_masked)) if len(pitches_masked) > 0 else 0.0
        pitch_std = float(np.std(pitches_masked))  if len(pitches_masked) > 0 else 0.0

        # ── 2. Tempo ─────────────────────────────────────────────────────────
        onset_env = librosa.onset.onset_strength(y=y, sr=sr)
        tempo_arr = librosa.beat.tempo(onset_envelope=onset_env, sr=sr)
        avg_tempo = float(tempo_arr[0]) if len(tempo_arr) > 0 else 0.0

        # ── 3. Energy ────────────────────────────────────────────────────────
        rms        = librosa.feature.rms(y=y)[0]
        avg_energy = float(np.mean(rms))
        energy_std = float(np.std(rms))

        # ── 4. Spectral features ─────────────────────────────────────────────
        avg_centroid = float(np.mean(librosa.feature.spectral_centroid(y=y, sr=sr)[0]))
        avg_flatness = float(np.mean(librosa.feature.spectral_flatness(y=y)[0]))
        avg_zcr      = float(np.mean(librosa.feature.zero_crossing_rate(y=y)[0]))

        # ── Objective English summary ─────────────────────────────────────────
        desc: list = []

        if avg_tempo > 130:
            desc.append("Fast speaking pace (high energy or nervous).")
        elif avg_tempo < 80:
            desc.append("Slow speaking pace (measured or low energy).")
        else:
            desc.append("Normal/moderate speaking pace.")

        if pitch_std > 60:
            desc.append("High pitch variation (expressive or excited delivery).")
        elif pitch_std < 20:
            desc.append("Low pitch variation (monotone or robotic delivery).")

        energy_ratio = energy_std / (avg_energy + 1e-6)
        if energy_ratio > 0.5:
            desc.append("Significant energy fluctuations (possible hesitation or shaky voice).")
        else:
            desc.append("Stable vocal energy (suggesting confidence).")

        if avg_flatness > 0.05:
            desc.append("Breathy or noisy vocal texture detected.")

        return {
            "summary": " ".join(desc),
            "stats": {
                "avg_pitch_hz":       avg_pitch,
                "pitch_std":          pitch_std,
                "tempo_bpm":          avg_tempo,
                "avg_energy":         avg_energy,
                "energy_fluctuation": energy_std,
                "avg_centroid":       avg_centroid,
                "avg_flatness":       avg_flatness,
                "avg_zcr":            avg_zcr,
            },
        }

    except Exception as e:
        logger.error(f"Audio analysis failed: {e}")
        return {"summary": "Technical audio extraction failed.", "stats": {}}


# ---------------------------------------------------------------------------
# Core analysis function
# ---------------------------------------------------------------------------

def analyze_response(
    question:          str,
    predefined_answer: str,
    user_answer:       str,
    user_voice_url:    Optional[str] = None,
    language:          Optional[str] = None,
) -> Dict[str, Any]:
    """
    Evaluates a candidate's answer against the ideal answer using qwen2.5:14b.

    Parameters
    ----------
    question           : The interview / assessment question.
    predefined_answer  : The ideal / expected answer.
    user_answer        : The candidate's actual answer.
    user_voice_url     : Optional URL or Base64 audio of the candidate's answer.
    language           : Language key — one of: english, hindi, marathi, tamil,
                         telugu, kannada, punjabi.  Defaults to 'english'.

    Returns
    -------
    dict with keys: matchScore, comparison, behavioralAnalysis, tonalAnalysis,
                    visuals, voiceTranscription (when voice provided)
    """

    lang_cfg = get_language_config(language)
    logger.info(
        f"Starting analysis | language={lang_cfg['display_name']} | model={OLLAMA_MODEL}"
    )

    # ------------------------------------------------------------------
    # 1. Audio processing — transcription + acoustic features
    # ------------------------------------------------------------------
    audio_features: Optional[Dict[str, Any]] = None
    voice_transcription: Optional[Dict[str, Any]] = None

    if user_voice_url:
        logger.info("Processing audio input...")
        audio_file_path = download_or_save_audio(user_voice_url)
        if audio_file_path:
            # Step A: Transcribe the speech using Whisper
            logger.info("Transcribing audio via Whisper...")
            voice_transcription = transcribe_audio(audio_file_path, language)

            # Step B: Extract acoustic features via librosa
            logger.info("Extracting acoustic features via librosa...")
            audio_features = analyze_audio_features(audio_file_path)

            try:
                os.remove(audio_file_path)
            except Exception:
                pass

    # ------------------------------------------------------------------
    # 2. System prompt — language-aware
    # ------------------------------------------------------------------
    system_prompt = (
        f"You are a world-class communication expert, behavioral psychologist, and tonal analyst "
        f"with deep expertise in {lang_cfg['script_name']}. "
        f"Your goal is to provide a deep, professional evaluation of a candidate's interview response. "
        f"Analyze the text for semantic accuracy AND the voice for emotional nuance when voice data is provided. "
        f"{lang_cfg['response_instruction']} "
        f"Return the result ONLY in strict JSON format — no markdown code fences, no preamble, no extra text."
    )

    # ------------------------------------------------------------------
    # 3. Audio context block — native-language labels + objective stats
    # ------------------------------------------------------------------
    audio_context_block = ""
    if audio_features:
        audio_context_block = f"""
    {lang_cfg['audio_context_header']}
    Objective Delivery Summary (English): {audio_features['summary']}
    Raw Acoustic Stats: {json.dumps(audio_features['stats'], ensure_ascii=False)}
    ({lang_cfg['audio_hint']})
    """

    # ------------------------------------------------------------------
    # 3b. Voice transcription block — what the user actually SAID
    # ------------------------------------------------------------------
    voice_transcription_block = ""
    speech_consistency_block = ""
    transcribed_text = ""

    if voice_transcription and voice_transcription.get("transcription"):
        transcribed_text = voice_transcription["transcription"]
        vt = voice_transcription  # shorthand

        voice_transcription_block = f"""
    ═══════════════════════════════════════════════════════════════
    VOICE TRANSCRIPTION (What the candidate actually SPOKE):
    ═══════════════════════════════════════════════════════════════
    Transcribed Text    : "{transcribed_text}"
    Detected Language   : {vt.get('detected_language', 'unknown')}
    Confidence Score    : {vt.get('confidence', 0)}
    Speaking Duration   : {vt.get('speaking_duration_sec', 0)} seconds
    Words Per Minute    : {vt.get('words_per_minute', 0)}
    Pause Count         : {vt.get('pause_count', 0)} (pauses > 1.5s)
    Total Pause Duration: {vt.get('total_pause_duration_sec', 0)} seconds
    """

        speech_consistency_block = f"""
    ═══════════════════════════════════════════════════════════════
    SPEECH-TEXT CONSISTENCY CHECK:
    ═══════════════════════════════════════════════════════════════
    The user typed this answer  : "{user_answer}"
    The user SPOKE this answer  : "{transcribed_text}"

    IMPORTANT: Compare the typed answer with the spoken answer above.
    If they differ significantly, this is a CRITICAL observation —
    the candidate may have typed one thing but said something different.
    Factor this into ALL analysis fields (comparison, behavioralAnalysis,
    tonalAnalysis). The SPOKEN answer should be treated as the PRIMARY
    source of truth for what the candidate actually communicated.
    """

    # ------------------------------------------------------------------
    # 4. User prompt — structured and language-aware
    # ------------------------------------------------------------------
    if voice_transcription and voice_transcription.get("transcription"):
        tonal_instruction = (
            f"You have ACTUAL VOICE DATA with both transcription and acoustic metrics. "
            f"Provide a comprehensive tonal analysis that includes: "
            f"(a) Vocal delivery assessment — confidence, pace, fluency, emotional tone "
            f"based on the acoustic stats (pitch variation, energy, tempo). "
            f"(b) Speech clarity — articulation quality, speaking rate ({voice_transcription.get('words_per_minute', 0)} WPM), "
            f"pauses ({voice_transcription.get('pause_count', 0)} detected). "
            f"(c) Emotional indicators — nervousness, confidence, hesitation based on "
            f"energy fluctuations and pause patterns. "
            f"(d) Speech-text alignment — note if spoken words differ from typed answer. "
            f"Write the ENTIRE analysis in {lang_cfg['display_name']}."
        )
    else:
        tonal_instruction = (
            f"If voice metadata IS provided above: give a detailed breakdown of the delivery "
            f"(confidence, pace, emotional stability) written in {lang_cfg['display_name']}. "
            f"{lang_cfg['text_tone_fallback']}"
        )

    # Determine which answer to emphasize for scoring
    answer_source_note = ""
    if transcribed_text:
        answer_source_note = (
            f"\n    NOTE: The candidate provided BOTH a typed answer and a voice answer. "
            f"The voice transcription is: \"{transcribed_text}\". "
            f"Use the SPOKEN answer as the PRIMARY basis for matchScore calculation, "
            f"since it represents what the candidate actually communicated verbally. "
            f"The typed answer should be used as supplementary context."
        )

    prompt = f"""
Evaluate the following interview response with high precision.

INPUT DATA:
- Content Language  : {lang_cfg['display_name']} ({lang_cfg['script_name']})
- Question          : "{question}"
- Ideal Answer      : "{predefined_answer}"
- User's Typed Answer : "{user_answer}"
{voice_transcription_block}{speech_consistency_block}{audio_context_block}{answer_source_note}

REQUIRED ANALYSIS — ALL text fields MUST be written in {lang_cfg['display_name']}:

1. matchScore (integer 0–100):
   Quantify how well the user's answer covers the core concepts of the ideal answer.
   {'When voice transcription is available, score primarily based on the SPOKEN answer.' if transcribed_text else ''}
   Be strict but fair. Deduct points for missing key ideas; award partial credit for partial coverage.

2. comparison (text in {lang_cfg['display_name']}):
   A professional 2–3 sentence summary explaining WHY the score was given.
   Highlight what was covered correctly and what key points were missing.
   {'If the spoken answer differs from the typed answer, explicitly mention this discrepancy and its impact on the score.' if transcribed_text else ''}
   {lang_cfg['response_instruction']}

3. behavioralAnalysis (text in {lang_cfg['display_name']}):
   Analyse the candidate's communication style:
   - Are they assertive, analytical, empathetic, or vague?
   - Is there logical flow and a coherent structure?
   - Comment on vocabulary richness and appropriateness for {lang_cfg['display_name']}.
   {'- Analyse speech-text consistency: did the candidate say what they typed, or were there notable differences?' if transcribed_text else ''}
   {'- Comment on verbal fluency: hesitations, filler words, speaking pace.' if transcribed_text else ''}
   {lang_cfg['response_instruction']}

4. tonalAnalysis (text in {lang_cfg['display_name']}):
   {tonal_instruction}
   {lang_cfg['response_instruction']}

5. visuals (array of exactly 3 items):
   {lang_cfg['visuals_instruction']}

OUTPUT FORMAT — strict JSON only, absolutely no markdown fences or extra text:
{{
    "matchScore": <integer 0-100>,
    "comparison": "<text in {lang_cfg['display_name']}>",
    "behavioralAnalysis": "<text in {lang_cfg['display_name']}>",
    "tonalAnalysis": "<text in {lang_cfg['display_name']}>",
    "visuals": ["<kw1>", "<kw2>", "<kw3>"]
}}
"""

    # ------------------------------------------------------------------
    # 5. Ollama API payload
    # ------------------------------------------------------------------
    payload = {
        "model": OLLAMA_MODEL,
        "prompt": prompt,
        "system": system_prompt,
        "stream": False,
        "format": "json",
        "options": {
            "temperature": 0.15,   # Low for consistent, professional output
            "num_ctx":     8192,   # qwen2.5:14b supports large context; utilise it
            "num_predict": 2048,   # Generous budget for verbose non-Latin scripts
        },
    }

    # ------------------------------------------------------------------
    # 6. Call the LLM
    # ------------------------------------------------------------------
    llm_response_text = ""

    try:
        logger.info(f"Calling Ollama API — model: {OLLAMA_MODEL}...")
        response = requests.post(OLLAMA_API_URL, json=payload, timeout=300)

        if response.status_code == 404:
            msg = (
                f"Model '{OLLAMA_MODEL}' not found on this Ollama instance. "
                f"Pull it with: ollama pull {OLLAMA_MODEL}"
            )
            logger.error(msg)
            return {"error": msg}

        response.raise_for_status()
        result = response.json()

        llm_response_text = result.get("response", "").strip()

        # Strip markdown fences if the model adds them despite our instructions
        if "```json" in llm_response_text:
            llm_response_text = (
                llm_response_text.split("```json")[1].split("```")[0].strip()
            )
        elif "```" in llm_response_text:
            llm_response_text = (
                llm_response_text.split("```")[1].split("```")[0].strip()
            )

        parsed = json.loads(llm_response_text)

        # ── Sanity-check required keys ────────────────────────────────────
        required_keys = {
            "matchScore", "comparison", "behavioralAnalysis",
            "tonalAnalysis", "visuals"
        }
        missing = required_keys - parsed.keys()
        if missing:
            logger.warning(f"LLM response is missing keys: {missing}. Filling with defaults.")
            for k in missing:
                parsed[k] = [] if k == "visuals" else ""

        # ── Clamp matchScore to 0–100 ─────────────────────────────────────
        try:
            parsed["matchScore"] = max(0, min(100, int(parsed["matchScore"])))
        except (ValueError, TypeError):
            parsed["matchScore"] = 0

        # ── Ensure visuals has exactly 3 items ────────────────────────────
        if not isinstance(parsed.get("visuals"), list):
            parsed["visuals"] = []
        while len(parsed["visuals"]) < 3:
            parsed["visuals"].append("")
        parsed["visuals"] = parsed["visuals"][:3]

        # ── Attach voice transcription to response if available ─────────
        if voice_transcription and voice_transcription.get("transcription"):
            parsed["voiceTranscription"] = voice_transcription["transcription"]
        else:
            parsed["voiceTranscription"] = None

        logger.info(
            f"Analysis complete | matchScore={parsed['matchScore']} | "
            f"language={lang_cfg['display_name']} | "
            f"has_transcription={'yes' if parsed.get('voiceTranscription') else 'no'}"
        )
        return parsed

    except json.JSONDecodeError:
        logger.error("Failed to parse JSON from LLM response.")
        return {
            "error": "LLM returned invalid JSON.",
            "raw_response": llm_response_text,
        }

    except requests.RequestException as e:
        logger.error(f"Ollama connection error: {e}")
        return {"error": f"Could not connect to Ollama: {str(e)}"}


# ---------------------------------------------------------------------------
# CLI entry point
# ---------------------------------------------------------------------------
if __name__ == "__main__":
    import argparse
    import sys

    parser = argparse.ArgumentParser(
        description="Response IQ — CLI Analysis Tool",
        formatter_class=argparse.RawTextHelpFormatter,
    )
    parser.add_argument("--question",   type=str, help="The interview question")
    parser.add_argument("--predefined", type=str, help="The ideal answer")
    parser.add_argument("--user",       type=str, help="The user's answer")
    parser.add_argument("--voice",      type=str, default=None,
                        help="URL or Base64-encoded audio of the user's answer")
    parser.add_argument(
        "--language",
        type=str,
        default=DEFAULT_LANGUAGE,
        choices=SUPPORTED_LANGUAGES,
        help=f"Language of the content. Supported: {SUPPORTED_LANGUAGES}",
    )
    parser.add_argument(
        "--json_input",
        type=str,
        help=(
            "Path to a JSON file OR a raw JSON string.\n"
            "Expected keys: question, predefinedAnswer, userAnswer, userVoice, language"
        ),
    )

    args = parser.parse_args()
    q, pa, ua, uv, lang = (
        args.question, args.predefined, args.user, args.voice, args.language
    )

    if args.json_input:
        try:
            src = args.json_input
            if os.path.isfile(src):
                with open(src, encoding="utf-8") as f:
                    data = json.load(f)
            else:
                data = json.loads(src)

            q    = data.get("question")
            pa   = data.get("predefinedAnswer")
            ua   = data.get("userAnswer")
            uv   = data.get("userVoice")
            lang = data.get("language", DEFAULT_LANGUAGE)
        except Exception as e:
            print(f"Error parsing --json_input: {e}")
            sys.exit(1)

    # Default Hindi demo if no input provided
    if not q or not pa or not ua:
        print("No input provided — running Hindi demo...\n")
        q    = "एडुरिगो क्या है?"
        pa   = (
            "एडुरिगो एक व्यापक लर्निंग और एंगेजमेंट प्लेटफ़ॉर्म है, जिसे संगठनों के लिए "
            "ट्रेनिंग, असेसमेंट, गेमिफिकेशन और एनालिटिक्स प्रदान करने के लिए डिज़ाइन किया गया है। "
            "यह व्यवसायों को कर्मचारियों, पार्टनर्स और ग्राहकों को इंटरैक्टिव और स्केलेबल "
            "लर्निंग सॉल्यूशन्स के माध्यम से अपस्किल करने में मदद करता है।"
        )
        ua   = (
            "एडुरिगो एक व्यापक लर्निंग और एंगेजमेंट प्लेटफ़ॉर्म है, जिसे संगठनों के लिए "
            "ट्रेनिंग, असेसमेंट, गेमिफिकेशन और एनालिटिक्स प्रदान करने के लिए डिज़ाइन किया गया है।"
        )
        uv   = None
        lang = "hindi"

    print("─" * 65)
    print(f"  Model       : {OLLAMA_MODEL}")
    print(f"  Language    : {lang}")
    print(f"  Question    : {q}")
    print(f"  User Answer : {ua[:80]}{'...' if len(ua) > 80 else ''}")
    if uv:
        print(f"  Voice       : {uv[:60]}...")
    print("─" * 65)

    result = analyze_response(q, pa, ua, uv, language=lang)

    print("\n── Analysis Result ──────────────────────────────────────────\n")
    print(json.dumps(result, indent=2, ensure_ascii=False))
