# import pytesseract
from PIL import Image
import os
from django.conf import settings
from typing import Optional

class OCRService:
    """
    Service for extracting text from image and PDF files.
    """
    
    @staticmethod
    def extract_text(file_path: str) -> str:
        """
        Extract text from an image or PDF.
        Currently uses pytesseract for images. 
        For PDF, we can use libraries like pdf2image to convert to images first.
        """
        file_ext = os.path.splitext(file_path)[1].lower()
        
        if file_ext in ['.jpg', '.jpeg', '.png']:
            return OCRService._extract_from_image(file_path)
        elif file_ext == '.pdf':
            return OCRService._extract_from_pdf(file_path)
        else:
            raise ValueError(f"Unsupported file format: {file_ext}")

    @staticmethod
    def _extract_from_image(file_path: str) -> str:
        """
        Extract text from an image file using pytesseract.
        """
        try:
            image = Image.open(file_path)
            # pytesseract.pytesseract.tesseract_cmd = r'C:\Program Files\Tesseract-OCR\tesseract.exe' # Example for Windows
            text = pytesseract.image_to_string(image)
            return text
        except Exception as e:
            # Fallback or log error
            print(f"OCR Error: {str(e)}")
            return ""

    @staticmethod
    def _extract_from_pdf(file_path: str) -> str:
        """
        Extract text from a PDF file.
        Simplified implementation using pypdf for text-based PDFs.
        For scanned PDFs, we should use pdf2image + OCR.
        """
        try:
            from pypdf import PdfReader
            reader = PdfReader(file_path)
            text = ""
            for page in reader.pages:
                text += page.extract_text() + "\n"
            
            # If no text extracted (possibly a scanned PDF), fallback to placeholder
            if not text.strip():
                return "OCR extraction from scanned PDF not fully implemented. Please use images."
            
            return text
        except Exception as e:
            print(f"PDF Extraction Error: {str(e)}")
            return ""
