# Copyright 2025 The HuggingFace Team. All rights reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
#     http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.

import logging
import os
import sys

import datasets
import transformers
from datasets import load_dataset
from transformers import set_seed
from transformers.trainer_utils import get_last_checkpoint

from grpo_configs_v2 import GRPOConfig, GRPOScriptArguments
from rewards_v2 import get_reward_funcs, set_judge_generator
from grpo_utils_v2 import get_model, get_tokenizer
from grpo_data_util import generate_tictactoe_prompt
# from utils.callbacks import get_callbacks
from grpo_utils_v2 import init_wandb_training
from trl import GRPOTrainer, ModelConfig, TrlParser, get_peft_config

import pandas as pd
from datasets import Dataset

from torch.distributed.elastic.multiprocessing.errors import record


logger = logging.getLogger(__name__)

from transformers import Trainer
import torch, os

def _load_rng_state_safe(self, checkpoint: str):
    if checkpoint is None:
        return
    rng_file = os.path.join(checkpoint, "rng_state.pth")
    if os.path.isfile(rng_file):
        # turn the safety switch off for this one file
        self._rng_state = torch.load(rng_file, map_location="cpu", weights_only=False)

Trainer._load_rng_state = _load_rng_state_safe


@record
def main(script_args, training_args, model_args):
    # Set seed for reproducibility
    set_seed(training_args.seed)

    ###############
    # Setup logging
    ###############
    logging.basicConfig(
        format="%(asctime)s - %(levelname)s - %(name)s - %(message)s",
        datefmt="%Y-%m-%d %H:%M:%S",
        handlers=[logging.StreamHandler(sys.stdout)],
    )
    log_level = training_args.get_process_log_level()
    logger.setLevel(log_level)
    datasets.utils.logging.set_verbosity(log_level)
    transformers.utils.logging.set_verbosity(log_level)
    transformers.utils.logging.enable_default_handler()
    transformers.utils.logging.enable_explicit_format()

    # Log on each process a small summary
    logger.warning(
        f"Process rank: {training_args.local_rank}, device: {training_args.device}, n_gpu: {training_args.n_gpu}"
        + f" distributed training: {bool(training_args.local_rank != -1)}, 16-bits training: {training_args.fp16}"
    )
    logger.info(f"Model parameters {model_args}")
    logger.info(f"Script parameters {script_args}")
    logger.info(f"Training parameters {training_args}")

    # Check for last checkpoint
    last_checkpoint = None
    output_dir = training_args.output_dir
    representation_mode = script_args.representation_mode
    model_name_or_path = model_args.model_name_or_path
    dataset_type = script_args.dataset_type
    
    # Support best move
    experiment_mode = script_args.experiment_mode
    
    # Sanitize model name_or_path to not contain / or \
    model_name = model_name_or_path
    # If loading from checkpoint then we extract the existing model name
    if "checkpoint" in model_name:
        save_folder_name = model_name.split("/")[-2]
        print(f"save_folder_name: {save_folder_name}")
    elif "updated" in model_name:
        # For llama where we save an updated model with special tokens
        save_folder_name = model_name.split("/")[-1]
        print(f"save_folder_name: {save_folder_name}")
    else:
        model_name = model_name.replace("/", "_").replace("\\", "_")
        logger.info(f"Model name: {model_name}")
        save_folder_name = f"{model_name}_{representation_mode}_{dataset_type}_{experiment_mode}"
    save_path = os.path.join(output_dir, save_folder_name)
    training_args.output_dir = save_path      # now every periodic checkpoint lands here
    os.makedirs(save_path, exist_ok=True)     # folder guaranteed to exist
    logger.info(f"Save path: {save_path}")
    
    if training_args.overwrite_output_dir and os.path.isdir(save_path):
        logger.info(f"Output directory ({save_path}) already exists and overwrite_output_dir is set to True.")
    if os.path.isdir(save_path):
        last_checkpoint = get_last_checkpoint(save_path)
    if last_checkpoint is not None and training_args.resume_from_checkpoint is None:
        logger.info(f"Checkpoint detected, resuming training at {last_checkpoint=}.")
    

    training_args.run_name = training_args.run_name or f"{model_name}_{representation_mode}_{dataset_type}_{experiment_mode}"

    # if "wandb" in training_args.report_to:
    #     init_wandb_training(training_args)

    # Load the dataset
    # raw_dataset = load_dataset(script_args.dataset_name, name=script_args.dataset_config)
    
    # pdf = pd.read_feather(script_args.dataset_name)          # ① load Feather
    # raw_dataset = Dataset.from_pandas(pdf, preserve_index=False)  # ② wrap as HF Dataset

    
    # dataset = raw_dataset.map(
    #     explode_qa_pairs,
    #     batched=True,
    #     # remove_columns=[c for c in raw_dataset["train"].column_names
    #     #                 if c not in {"identifier", "past_timeline", "qa_pairs"}],  # keeps it tidy
    # ).with_format("python")     # we’ll still switch back to Arrow after tokenisation
    
    # ------------------------------------------------------------------
    dataset_type = script_args.dataset_type
    if dataset_type == "random_80_10_10":
        TRAIN_PATH="/mnt/shared/data/stlm-logic/datasets/random_train_dataset_0.8_0.1_0.1.json"
        VAL_PATH="/mnt/shared/data/stlm-logic/datasets/random_val_dataset_0.8_0.1_0.1.json"
        TEST_PATH="/mnt/shared/data/stlm-logic/datasets/random_test_dataset_0.8_0.1_0.1.json"
    elif dataset_type == "canconical-symmetry-grouping":
        TRAIN_PATH="/mnt/shared/data/stlm-logic/datasets/tictactoe_train.json"
        VAL_PATH="/mnt/shared/data/stlm-logic/datasets/tictactoe_val.json"
        TEST_PATH="/mnt/shared/data/stlm-logic/datasets/tictactoe_test.json"
    else:
        raise ValueError(f"Unknown dataset type: {dataset_type}")
        
    logging.info("*** Loading dataset ***")
    # 1️⃣ # Load your tic-tac-toe dataset.
    train_dataset = load_dataset("json",data_files=TRAIN_PATH, split="train")
    val_dataset = load_dataset("json",data_files=VAL_PATH, split="train")
    


    ################
    # Load tokenizer
    ################
    tokenizer = get_tokenizer(model_args, training_args)
    
    # tokenizer has no pad token set eos token as pad token
    if tokenizer.pad_token is None:
        tokenizer.pad_token = tokenizer.eos_token
        logger.info(f"Tokenizer pad token set to {tokenizer.pad_token}")
    
    # If using special move tokens, add them to the tokenizer's vocabulary using the new method.
    # if script_args.representation_mode == "special":
    #     special_tokens = [f"<move_{i}>" for i in range(1, 19)]
    #     # Check if tokens are already in the vocabulary.
    #     vocab = tokenizer.get_vocab()
    #     new_tokens = list(set(special_tokens) - set(vocab.keys()))
    #     if new_tokens:
    #         num_added_tokens = tokenizer.add_tokens(new_tokens)
    #         logger.info(f"Added {num_added_tokens} new tokens: {new_tokens}")
    #     else:
    #         logger.info("All special tokens already exist in the vocabulary.")
            


    ##############
    # Load model #
    ##############
    logger.info("*** Loading model ***")
    model = get_model(model_args, training_args)
    
    # Resize the model's token embeddings to match the new vocabulary size.
    if script_args.representation_mode == "special":
        if "Qwen" in model_args.model_name_or_path:
            model.config.vocab_size = len(tokenizer)
    else:
        logger.info("No resizing of model's token embeddings needed.")

    # Get reward functions from the registry
    reward_funcs = get_reward_funcs(script_args)

    # Format into conversation
    def make_conversation(example):
        """
        Produces:
            ─ 'prompt'   : list[{"role": "...", "content": "..."}]   (what GRPOTrainer expects)
            ─ 'allowed_moves': list[...]
        """
        processed_output = generate_tictactoe_prompt(
            example, 
            script_args.representation_mode,
            script_args.instruct_model,
            script_args.experiment_mode
        )
        prompt_text = processed_output["prompt"]
        allowed_moves = processed_output["allowed_moves"]

        messages = []
        if training_args.system_prompt:
            messages.append({"role": "system", "content": training_args.system_prompt})
        messages.append({"role": "user",   "content": prompt_text})

        return {"prompt": messages, "allowed_moves": allowed_moves}

    train_dataset = train_dataset.map(make_conversation)
    val_dataset = val_dataset.map(make_conversation)
    
    # log the first allowed_moves value
    logging.warning(f"First allowed_moves after make_conversation: {train_dataset[0]['allowed_moves']}")

    # for split in ["train", "val"]:
    #     if "messages" in eval(f"{split}_dataset").column_names:
    #         eval(f"{split}_dataset") = eval(f"{split}_dataset").remove_columns("messages")

    reward_processing_classes = [tokenizer for _ in reward_funcs]
    #############################
    # Initialize the GRPO trainer
    #############################
    trainer = GRPOTrainer(
        model=model,
        reward_funcs=reward_funcs,
        args=training_args,
        train_dataset=train_dataset,
        eval_dataset=val_dataset,
        peft_config=get_peft_config(model_args),
        # callbacks=get_callbacks(training_args, model_args),
        processing_class=tokenizer,
        reward_processing_classes=reward_processing_classes,
    )

    ###############
    # Training loop
    ###############
    logger.info("*** Train ***")
    checkpoint = None
    if training_args.resume_from_checkpoint is not None:
        checkpoint = training_args.resume_from_checkpoint
    elif last_checkpoint is not None:
        checkpoint = last_checkpoint
    train_result = trainer.train(resume_from_checkpoint=checkpoint)
    metrics = train_result.metrics
    metrics["train_samples"] = len(train_dataset)
    # metrics["val_samples"] = len(val_dataset)
    trainer.log_metrics("train", metrics)
    trainer.save_metrics("train", metrics)
    trainer.save_state()

    ##################################
    # Save model and create model card
    ##################################
    logger.info("*** Save model ***")
    trainer.save_model(save_path)
    logger.info(f"Model saved to {save_path}")

    # Save everything else on main process
    kwargs = {
        "dataset_name": script_args.dataset_name,
        "tags": ["open-r1"],
    }
    if trainer.accelerator.is_main_process:
        trainer.create_model_card(**kwargs)
        # Restore k,v cache for fast inference
        trainer.model.config.use_cache = True
        trainer.model.config.save_pretrained(save_path)

    ##########
    # Evaluate
    ##########
    if training_args.do_eval:
        logger.info("*** Evaluate ***")
        metrics = trainer.evaluate()
        metrics["eval_samples"] = len(val_dataset)
        trainer.log_metrics("eval", metrics)
        trainer.save_metrics("eval", metrics)

    #############
    # push to hub
    #############
    if training_args.push_to_hub:
        logger.info("Pushing to hub...")
        trainer.push_to_hub(**kwargs)


if __name__ == "__main__":
    parser = TrlParser((GRPOScriptArguments, GRPOConfig, ModelConfig))
    script_args, training_args, model_args = parser.parse_args_and_config()
    main(script_args, training_args, model_args)