# Copyright 2024 Bytedance Ltd. and/or its affiliates
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
#     http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.

import math
import os
from collections import defaultdict
from io import BytesIO
from typing import Any, Dict, List, Optional, Tuple, Union

import numpy as np
import torch
from datasets import (
    concatenate_datasets,
    get_dataset_config_names,
    load_dataset,
)
from jinja2 import Template
from PIL import Image
from PIL.Image import Image as ImageObject
from qwen_vl_utils.vision_process import fetch_video
from torch.utils.data import Dataset
from transformers import PreTrainedTokenizer, ProcessorMixin
import re

from ..models.transformers.qwen2_vl import get_rope_index
from . import torch_functional as VF


def collate_fn(features: List[Dict[str, Any]]) -> Dict[str, Any]:
    tensors = defaultdict(list)
    non_tensors = defaultdict(list)
    for feature in features:
        for key, value in feature.items():
            if isinstance(value, torch.Tensor):
                tensors[key].append(value)
            else:
                non_tensors[key].append(value)

    for key, value in tensors.items():
        tensors[key] = torch.stack(value, dim=0)

    for key, value in non_tensors.items():
        non_tensors[key] = np.array(value, dtype=object)

    return {**tensors, **non_tensors}


def process_image(
    image: Union[Dict[str, Any], ImageObject, str], min_pixels: Optional[int], max_pixels: Optional[int]
) -> ImageObject:
    if isinstance(image, str):
        image = Image.open(image)
    elif isinstance(image, dict):
        image = Image.open(BytesIO(image["bytes"]))
    elif isinstance(image, bytes):
        image = Image.open(BytesIO(image))

    image.load()  # avoid "Too many open files" errors
    if max_pixels is not None and (image.width * image.height) > max_pixels:
        resize_factor = math.sqrt(max_pixels / (image.width * image.height))
        width, height = int(image.width * resize_factor), int(image.height * resize_factor)
        image = image.resize((width, height))

    if min_pixels is not None and (image.width * image.height) < min_pixels:
        resize_factor = math.sqrt(min_pixels / (image.width * image.height))
        width, height = int(image.width * resize_factor), int(image.height * resize_factor)
        image = image.resize((width, height))

    if image.mode != "RGB":
        image = image.convert("RGB")

    return image


def process_video(
    video: str, min_pixels: Optional[int], max_pixels: Optional[int], video_fps: float, return_fps: bool = False
) -> Union[List[ImageObject], Tuple[List[ImageObject], List[float]]]:
    vision_info = {"video": video, "min_pixels": min_pixels, "max_pixels": max_pixels, "fps": video_fps}
    return fetch_video(vision_info, return_video_sample_fps=return_fps)


class RLHFDataset(Dataset):
    """
    We assume the dataset contains a column that contains prompts and other information
    """

    def __init__(
        self,
        data_path: str,
        tokenizer: PreTrainedTokenizer,
        processor: Optional[ProcessorMixin],
        prompt_key: str = "prompt",
        answer_key: str = "answer",
        image_key: str = "images",
        video_key: str = "videos",
        image_dir: Optional[str] = None,
        video_fps: float = 2.0,
        max_prompt_length: int = 6144,
        truncation: str = "error",
        format_prompt: Optional[str] = None,
        min_pixels: Optional[int] = None,
        max_pixels: Optional[int] = None,
        filter_overlong_prompts: bool = True,
        filter_overlong_prompts_workers: int = 16,
        insert_ground_truth: bool = False,
    ):
        self.tokenizer = tokenizer
        self.processor = processor
        self.prompt_key = prompt_key
        self.answer_key = answer_key
        self.image_key = image_key
        self.video_key = video_key
        self.image_dir = image_dir
        self.video_fps = video_fps
        self.max_prompt_length = max_prompt_length
        self.truncation = truncation
        self.min_pixels = min_pixels
        self.max_pixels = max_pixels
        
        self.insert_ground_truth = insert_ground_truth
            
        if "@" in data_path:
            data_path, _ = data_path.split("@")

            target_split = "validation"

            config_names = get_dataset_config_names(data_path)

            datasets_list = []
            for cfg in config_names:
                    ds = load_dataset(data_path, cfg, split=target_split)
                    if len(ds) > 0:
                        datasets_list.append(ds)

            if not datasets_list:
                raise RuntimeError(f"No datasets loaded for {data_path} (configs tried: {config_names})")

            self.dataset = concatenate_datasets(datasets_list)

        # self.format_prompt = None
        # if format_prompt:
        #     with open(format_prompt, encoding="utf-8") as f:
        #         self.format_prompt = f.read()
        
        self.user_prompt = """Your task is to answer the following multiple-choice question based on the provided image(s).

        Inside the <think> tag, you must demonstrate a careful and reflective thought process following these steps:
        1.  **Analyze the Question and Image(s)**: Break down the question, identify key information in the image(s), and determine what is being asked.
        2.  **Evaluate Each Option**: Systematically review each option. Provide reasoning for why an option is plausible or incorrect, referencing specific visual evidence and your knowledge.
        3.  **Critical Review and Final Conclusion**: Compare the options based on your analysis. State your final choice and provide a confident justification for why it is the best answer.

        After your thinking process, provide the final answer inside the <answer> tag. The answer must be **only the letter** of the correct option (e.g., A, B, C, D).

        **Question:**
        {Question}

        **Options:**
        {Options}

        <think>
        [Your thought process here, clearly showing the three steps above.]
        </think>
        <answer>
        {Answer}
        </answer>"""

        self.dataset = self.dataset.filter(
            self._filter_overlong_prompts,
            desc="Filtering overlong prompts",
            num_proc=16,
        )

    def _build_messages(self, example: Dict[str, Any]) -> List[Dict[str, Any]]:
        prompt_str: str = example[self.prompt_key]
        # if self.format_prompt:
        #     format_prompt = Template(self.format_prompt.strip())
        #     prompt_str = format_prompt.render(content=prompt_str)
        prompt_str = self.user_prompt.format(
                Question=prompt_str.lower().strip("."),
                Options=str(example["options"]),
                Answer="A")
        prompt_str = prompt_str.strip()

        content_list = []
        for i, content in enumerate(re.split(r"<image [1-7]>", prompt_str)):
            if i != 0:
                content_list.append({"type": "image"})

            if content:
                content_list.append({"type": "text", "text": content})

        return [{"role": "user", "content": content_list}]

    def __len__(self):
        return len(self.dataset)

    def _filter_overlong_prompts(self, example: Dict[str, Any]) -> bool:
        messages = self._build_messages(example)

        prompt = self.processor.apply_chat_template(messages, add_generation_prompt=True, tokenize=False)
        images = []
        total_img = 0
        for idx_img in range(1, 8):
            img = example["image_"+str(idx_img)]
            if img is not None and ("<image " + str(idx_img) + ">") in f"{example['question']} {str(example['options'])}":
                total_img += 1
                images.append(img)
        if total_img == 0:
            raise RuntimeError("No images found in the example.")
        
        processed_images = [] if len(images) != 0 else None  # text-only data
        for image in images:
            height, width = image.height, image.width
            if height < 28 or width < 28:
                # print(f"Image is too small: {height}x{width}")
                return False
            processed_images.append(process_image(image, self.min_pixels, self.max_pixels))

        model_inputs = self.processor(processed_images, [prompt], add_special_tokens=False, return_tensors="pt")

        size = model_inputs["input_ids"].size(-1)
        del model_inputs

        if len(processed_images) >= 3:
            return False
        else:
            return True
        
        return size <= self.max_prompt_length

    def __getitem__(self, index):
        example: dict = self.dataset[index]
            
        messages = self._build_messages(example)

        prompt = self.processor.apply_chat_template(messages, add_generation_prompt=True, tokenize=False)
        images = []
        total_img = 0
        for idx_img in range(1, 8):
            img = example["image_"+str(idx_img)]
            if img is not None and ("<image " + str(idx_img) + ">") in f"{example['question']} {str(example['options'])}":
                total_img += 1
                images.append(img)
        if total_img == 0:
            raise RuntimeError("No images found in the example.")
        
        processed_images = [] if len(images) != 0 else None  # text-only data
        for image in images:
            processed_images.append(process_image(image, self.min_pixels, self.max_pixels))

        model_inputs = self.processor(processed_images, [prompt], add_special_tokens=False, return_tensors="pt")
        input_ids = model_inputs.pop("input_ids")[0]
        attention_mask = model_inputs.pop("attention_mask")[0]
        example["multi_modal_data"] = {"images": images}

        if self.processor is not None and "Qwen2VLImageProcessor" in self.processor.image_processor.__class__.__name__:
            # qwen2vl mrope
            position_ids = get_rope_index(
                self.processor,
                input_ids=input_ids,
                image_grid_thw=model_inputs.get("image_grid_thw", None),
                video_grid_thw=model_inputs.get("video_grid_thw", None),
                second_per_grid_ts=model_inputs.get("second_per_grid_ts", None),
                attention_mask=attention_mask,
            )  # (3, seq_length)
        else:
            position_ids = torch.clip(attention_mask.cumsum(dim=0) - 1, min=0, max=None)  # (seq_length,)

        input_ids, attention_mask, position_ids = VF.postprocess_data(
            input_ids=input_ids,
            attention_mask=attention_mask,
            position_ids=position_ids,
            max_length=self.max_prompt_length,
            pad_token_id=self.tokenizer.pad_token_id,
            left_pad=True,
            truncation=self.truncation,
        )
        raw_prompt_ids = self.tokenizer.encode(prompt, add_special_tokens=False)
        if len(raw_prompt_ids) > self.max_prompt_length:
            if self.truncation == "left":
                raw_prompt_ids = raw_prompt_ids[-self.max_prompt_length :]
            elif self.truncation == "right":
                raw_prompt_ids = raw_prompt_ids[: self.max_prompt_length]
            elif self.truncation == "error":
                raise RuntimeError(f"Prompt length {len(raw_prompt_ids)} is longer than {self.max_prompt_length}.")

        example["input_ids"] = input_ids
        example["attention_mask"] = attention_mask
        example["position_ids"] = position_ids
        example["raw_prompt_ids"] = raw_prompt_ids
        example["ground_truth"] = example.pop(self.answer_key)
        
        if self.insert_ground_truth:
            ground_truth_text = example["ground_truth"]
            example["ground_truth_ids"] = self.tokenizer.encode(f"<answer>{ground_truth_text}</answer><|im_end|>", 
                                                                add_special_tokens=False)[: self.max_prompt_length//2] # different EOS for different models
        
        with open("debug_prompt.txt", "a", encoding="utf-8") as f:
            f.write(f"=== Example {index} ===\n")
            f.write(f"Prompt:\n{prompt}\n")
            f.write(f"Ground Truth:\n{example['ground_truth']}\n\n")
            f.write(f"Prompt:{prompt}\n\n")
            f.write(f"Total length of processed images: {len(processed_images)}\n")
        
        if(prompt.count("<|vision_start|><|image_pad|><|vision_end|>") != total_img):
            raise ValueError("WARNING: Mismatch in number of images processed.")
        return example
