import os
import json
import logging
import re
from typing import List, Dict, Any, Optional
from django.conf import settings
import openai
from google import genai
from google.genai import types
from settings.models import AppSettings

logger = logging.getLogger(__name__)

class AIParserService:
    """
    Service for parsing routine text using LLM (OpenAI or Gemini).
    """
    
    @staticmethod
    def ai_parse_service(text: str) -> List[Dict[str, Any]]:
        """
        Switch between providers based on database settings.
        """
        use_gemini = AppSettings.get_setting('use_gemini_ai', True)
        use_openai = AppSettings.get_setting('use_openai_ai', False)
        
        if use_gemini:
            return AIParserService._parse_with_gemini(text)
        elif use_openai:
            return AIParserService._parse_with_openai(text)
        else:
            raise ValueError("No AI provider enabled. Please enable either Gemini or OpenAI in AppSettings.")

    @staticmethod
    def _parse_with_openai(text: str) -> List[Dict[str, Any]]:
        """
        Send OCR text to OpenAI and get structured JSON response.
        """
        api_key = getattr(settings, 'OPENAI_API_KEY', os.environ.get('OPENAI_API_KEY'))
        
        if not api_key:
            logger.error("OpenAI API key not found in settings or environment.")
            raise ValueError("OpenAI API key is required for AI parsing.")

        client = openai.OpenAI(api_key=api_key)
        
        prompt = AIParserService._get_prompt(text)

        try:
            response = client.chat.completions.create(
                model="gpt-3.5-turbo", # or gpt-4o
                messages=[
                    {"role": "system", "content": "You are a precise data extraction assistant."},
                    {"role": "user", "content": prompt}
                ],
                temperature=0,
                timeout=30
            )
            
            content = response.choices[0].message.content.strip()
            return AIParserService._clean_and_parse_json(content)
            
        except Exception as e:
            logger.error(f"OpenAI Parsing Error: {str(e)}")
            raise

    @staticmethod
    def _parse_with_gemini(text: str) -> List[Dict[str, Any]]:
        """
        Send OCR text to Gemini and get structured JSON response using the new google-genai SDK.
        """
        api_key = getattr(settings, 'GEMINI_API_KEY', os.environ.get('GEMINI_API_KEY'))
        
        if not api_key:
            logger.error("Gemini API key not found in settings or environment.")
            raise ValueError("Gemini API key is required for AI parsing.")

        client = genai.Client(api_key=api_key)
        
        prompt = AIParserService._get_prompt(text)

        try:
            response = client.models.generate_content(
                model='google/gemma-3-27b-it',
                contents=prompt,
                config=types.GenerateContentConfig(
                    temperature=0,
                    response_mime_type="application/json",
                )
            )

            content = response.text.strip()
            return AIParserService._clean_and_parse_json(content)
            
        except Exception as e:
            logger.error(f"Gemini Parsing Error: {str(e)}")
            raise

    @staticmethod
    def _get_prompt(text: str) -> str:
        return """
                Act as a precise data extraction specialist. Convert the provided university class routine into the exact JSON structure specified below.

                ### EXTRACTION RULES:
                1. HEADER: Extract 'batch', 'department', 'year', and 'semester' from the top lines of the text.
                2. DYNAMIC TIME SLOTS: 
                - Parse all time intervals from the "Time" row.
                - For each interval, set "type" to "class" UNLESS the column is marked as a break.
                - For breaks, set "type" to "break" and "title" to the specific text found (e.g., "Tea Break" or "Lunch Break").
                3. TEACHERS: 
                - Extract the teacher list from the bottom of the text.
                - Split names into 'first_name' and 'last_name' (place middle names/initials in 'first_name').
                - Include the unique teacher "code" (e.g., "NH", "SI").
                4. COURSES: Map 'code', 'name', and 'credit' from the course table.
                5. ROUTINE DATA:
                - Map each day to an array of class objects.
                - Use 'slot' as the 0-based index corresponding to the 'time_slots' array.
                - IMPORTANT: If a course cell spans across multiple time slots in the text (e.g., sessions or double periods), you MUST create a separate object for each slot index.
                - Include 'room' and any notes like '(Every Alternative Week)'.

                ### JSON SCHEMA:
                {{
                "name": "1st Batch_32.pdf",
                "routine_year": "2026",
                "effective_from": "01-04-2026",
                "batch": "",
                "department": "",
                "semester": "",
                "year": "",
                "teachers": [{{"first_name": "", "last_name": "", "code": ""}}],
                "courses": [{{"code": "", "name": "", "credit": ""}}],
                "time_slots": [{{"start": "", "end": "", "type": "", "title": ""}}],
                "routine_data": {{
                    "Monday": [{{"slot": 0, "course": "", "teacher": "", "room": ""}}],
                    "Tuesday": [], "Wednesday": [], "Thursday": [], "Friday": []
                }}
                }}

                ### INPUT TEXT:
                {text}
                
                ### FINAL INSTRUCTION:
                Return ONLY the JSON object. Do not include any markdown styling (like ```json) or introductory text.
                """.format(text=text)

    @staticmethod
    def _clean_and_parse_json(content: str) -> List[Dict[str, Any]]:
        """
        Extract and parse JSON array from LLM response.
        Handles cases where LLM might include markdown formatting or extra text.
        """
        try:
            # Try parsing directly first
            return json.loads(content)
        except json.JSONDecodeError:
            # Try to find JSON array using regex
            match = re.search(r'\[\s*\{.*\}\s*\]', content, re.DOTALL)
            if match:
                try:
                    return json.loads(match.group())
                except json.JSONDecodeError as e:
                    logger.error(f"Failed to parse extracted JSON: {str(e)}")
                    raise ValueError("Malformed JSON received from AI.")
            else:
                logger.error(f"No JSON array found in AI response: {content}")
                raise ValueError("No valid routine data found by AI.")
