import io
import os
import logging
import pytesseract
from PIL import Image
import fitz

# Disable Pillow limit for high-resolution RERA documents to avoid DecompressionBombWarning
Image.MAX_IMAGE_PIXELS = None

logger = logging.getLogger(__name__)

class PDFContentExtractor:
    def __init__(self):
        self._initialize_ocr_engine()

    def _initialize_ocr_engine(self):
        common_windows_paths = [
            r'C:\Program Files\Tesseract-OCR\tesseract.exe',
            r'C:\Users\webbrains\AppData\Local\Programs\Tesseract-OCR\tesseract.exe',
        ]
        
        try:
            pytesseract.get_tesseract_version()
        except pytesseract.TesseractNotFoundError:
            for path in common_windows_paths:
                if os.path.exists(path):
                    pytesseract.pytesseract.tesseract_cmd = path
                    return
            logger.error("OCR engine (Tesseract) not found.")

    def extract_text(self, file_path: str) -> str:
        if not os.path.exists(file_path):
            return ""

        extracted_pages = []
        try:
            with fitz.open(file_path) as document:
                for page in document:
                    # 1. Try native text extraction
                    native_text = page.get_text("text").strip()
                    
                    # 2. Hybrid Check: If native text is too short, the page might be an image or have complex fonts
                    # 50 characters is a safe threshold for a meaningful page
                    if len(native_text) < 50:
                        logger.debug(f"Low text density ({len(native_text)} chars). Using high-res OCR fallback.")
                        # 300 DPI is standard for professional OCR (Gujarati/Hindi need clarity)
                        pix = page.get_pixmap(dpi=300) 
                        img = Image.open(io.BytesIO(pix.tobytes("png")))
                        # Extracting English, Hindi, and Gujarati
                        page_text = pytesseract.image_to_string(img, lang="eng+hin+guj").strip()
                    else:
                        page_text = native_text
                    
                    extracted_pages.append(page_text)
        except Exception as error:
            logger.error(f"Extraction error: {error}")
            
        return "\n".join(extracted_pages)

extractor = PDFContentExtractor()

def extract_content(pdf_path: str) -> str:
    return extractor.extract_text(pdf_path)
