import tiktoken
from typing import List
from app.core.config import settings
from app.core.logger import logger


class Tokenizer:
    """Handle token counting and text segmentation for embedding models."""

    def __init__(self, encoding_name: str = settings.TOKENIZER_MODEL):
        try:
            self.encoding = tiktoken.get_encoding(encoding_name)
            self.encoding_name = encoding_name
        except Exception as e:
            logger.error(f"Failed to load tokenizer {encoding_name}: {e}")
            # Fallback to cl100k_base
            self.encoding = tiktoken.get_encoding("cl100k_base")
            self.encoding_name = "cl100k_base"

    def count_tokens(self, text: str) -> int:
        """Count tokens in text."""
        try:
            tokens = self.encoding.encode(text)
            return len(tokens)
        except Exception as e:
            logger.error(f"Error counting tokens: {e}")
            return 0

    def count_tokens_batch(self, texts: List[str]) -> List[int]:
        """Count tokens for multiple texts."""
        return [self.count_tokens(text) for text in texts]

    def truncate_text(self, text: str, max_tokens: int) -> str:
        """Truncate text to fit within max token limit."""
        try:
            tokens = self.encoding.encode(text)
            if len(tokens) <= max_tokens:
                return text
            
            truncated_tokens = tokens[:max_tokens]
            return self.encoding.decode(truncated_tokens)
        except Exception as e:
            logger.error(f"Error truncating text: {e}")
            return text[:len(text) // 4 * max_tokens]  # Rough fallback

    def split_by_tokens(
        self, text: str, chunk_size: int, overlap: int = 0
    ) -> List[str]:
        """Split text into chunks by token count."""
        try:
            tokens = self.encoding.encode(text)
            chunks = []
            
            for i in range(0, len(tokens), chunk_size - overlap):
                chunk_tokens = tokens[i : i + chunk_size]
                chunk_text = self.encoding.decode(chunk_tokens)
                chunks.append(chunk_text)
            
            return chunks
        except Exception as e:
            logger.error(f"Error splitting by tokens: {e}")
            return [text]


# Singleton instance
_tokenizer_instance = None


def get_tokenizer() -> Tokenizer:
    """Get or create tokenizer instance."""
    global _tokenizer_instance
    if _tokenizer_instance is None:
        _tokenizer_instance = Tokenizer()
    return _tokenizer_instance
