import torch
import numpy as np
from onpolicy.utils.util import get_shape_from_obs_space, get_shape_from_act_space


def _flatten(T, N, x):
    return x.reshape(T * N, *x.shape[2:])


def _cast(x):
    return x.transpose(1, 2, 0, 3).reshape(-1, *x.shape[3:])


class SharedReplayBuffer(object):
    """
    Buffer to store training data.
    :param args: (argparse.Namespace) arguments containing relevant model, policy, and env information.
    :param num_agents: (int) number of agents in the env.
    :param obs_space: (gym.Space) observation space of agents.
    :param cent_obs_space: (gym.Space) centralized observation space of agents.
    :param act_space: (gym.Space) action space for agents.
    """

    def __init__(self, args, num_agents, obs_space, cent_obs_space, act_space):
        self.episode_length = args.episode_length
        self.n_rollout_threads = args.n_rollout_threads
        self.hidden_size = args.hidden_size
        self.recurrent_N = args.recurrent_N
        self.gamma = args.gamma
        self.gae_lambda = args.gae_lambda
        self._use_gae = args.use_gae
        self._use_popart = args.use_popart
        self._use_proper_time_limits = args.use_proper_time_limits
        self.args = args

        obs_shape = get_shape_from_obs_space(obs_space)
        share_obs_shape = get_shape_from_obs_space(cent_obs_space)

        if type(obs_shape[-1]) == list:
            obs_shape = obs_shape[:1]

        if type(share_obs_shape[-1]) == list:
            share_obs_shape = share_obs_shape[:1]

        self.share_obs = np.zeros((self.episode_length + 1, self.n_rollout_threads, num_agents, *share_obs_shape),
                                  dtype=np.float32)
        self.obs = np.zeros((self.episode_length + 1, self.n_rollout_threads, num_agents, *obs_shape), dtype=np.float32)

        self.rnn_states = np.zeros(
            (self.episode_length + 1, self.n_rollout_threads, num_agents, self.recurrent_N, self.hidden_size),
            dtype=np.float32)
        self.rnn_states_critic = np.zeros_like(self.rnn_states)


        self.value_preds = np.zeros(
            (self.episode_length + 1, self.n_rollout_threads, num_agents, 1), dtype=np.float32)

        self.returns = np.zeros_like(self.value_preds)
        if self.args.use_q or self.args.sp_use_q:
            self.target_rnn_states_critic = np.zeros_like(self.rnn_states)
            self.value_curr = np.zeros(
                (self.episode_length + 1, self.n_rollout_threads, num_agents, 1), dtype=np.float32)
            self.curr_returns = np.zeros_like(self.value_preds)
            self.baseline = np.zeros(
                (self.episode_length + 1, self.n_rollout_threads, num_agents, 1), dtype=np.float32)
            self.old_probs_all = np.zeros(
                (self.episode_length + 1, self.n_rollout_threads, num_agents, self.args.n_actions), dtype=np.float32)
            self.old_values_all = np.zeros(
                (self.episode_length + 1, self.n_rollout_threads, num_agents, self.args.n_actions), dtype=np.float32)
        if self.args.penalty_method and not self.args.env_name == 'mujoco':
            self.penalty_old_probs_all = np.zeros(
                (self.episode_length + 1, self.n_rollout_threads, num_agents, self.args.n_actions), dtype=np.float32)

        if self.args.sp_clip:
            self.sp_value_all = np.zeros(
                (self.episode_length + 1, self.n_rollout_threads, num_agents, self.args.sp_num), dtype=np.float32)
            self.sp_prob_all = np.zeros(
                (self.episode_length + 1, self.n_rollout_threads, num_agents, self.args.sp_num), dtype=np.float32)

        if act_space.__class__.__name__ == 'Discrete':
            print('buffer_discrete')
            self.available_actions = np.ones((self.episode_length + 1, self.n_rollout_threads, num_agents, act_space.n),
                                             dtype=np.float32)
        else:
            print('buffer_continue')
            self.available_actions = None

        act_shape = get_shape_from_act_space(act_space)
        # print('act_shape = {}'.format(act_shape))
        self.actions = np.zeros(
            (self.episode_length, self.n_rollout_threads, num_agents, act_shape), dtype=np.float32)
        self.action_log_probs = np.zeros(
            (self.episode_length, self.n_rollout_threads, num_agents, act_shape), dtype=np.float32)
        self.rewards = np.zeros(
            (self.episode_length, self.n_rollout_threads, num_agents, 1), dtype=np.float32)

        self.masks = np.ones((self.episode_length + 1, self.n_rollout_threads, num_agents, 1), dtype=np.float32)
        self.bad_masks = np.ones_like(self.masks)
        self.active_masks = np.ones_like(self.masks)

        self.step = 0

        self.n_agents = num_agents
        self.update_index = 0
        self.total_update_step = 0
        if self.args.aga_F < 1:
            F_total =  int(1 / self.args.aga_F)
            F_iql = 1
        else:
            F_total = int(self.args.aga_F)
            F_iql = F_total - 1
        if self.args.new_period:
            self.total_period =  self.n_agents * F_total
            self.iql_period = self.n_agents * F_iql
        else:
            self.total_period = self.args.period * self.n_agents * F_total
            self.iql_period = self.args.period * self.n_agents * F_iql
        if self.args.aga_first:
            self.aga_update_tag = (self.total_update_step % self.total_period) <= (self.total_period - self.iql_period)
        else:
            self.aga_update_tag = (self.total_update_step % self.total_period) >= self.iql_period

    def insert(self, share_obs, obs, rnn_states_actor, rnn_states_critic, actions, action_log_probs,
               value_preds, rewards, masks, bad_masks=None, active_masks=None, available_actions=None,insert_mode ='multi',\
               curr_rollout=None,target_rnn_states_critic=None,value_curr=None,baseline=None,old_probs_all=None,old_values_all=None,\
               sp_value=None,sp_prob=None,penalty_old_probs=None):
        """
        Insert data into the buffer.
        :param share_obs: (argparse.Namespace) arguments containing relevant model, policy, and env information.
        :param obs: (np.ndarray) local agent observations.
        :param rnn_states_actor: (np.ndarray) RNN states for actor network.
        :param rnn_states_critic: (np.ndarray) RNN states for critic network.
        :param actions:(np.ndarray) actions taken by agents.
        :param action_log_probs:(np.ndarray) log probs of actions taken by agents
        :param value_preds: (np.ndarray) value function prediction at each step.
        :param rewards: (np.ndarray) reward collected at each step.
        :param masks: (np.ndarray) denotes whether the environment has terminated or not.
        :param bad_masks: (np.ndarray) action space for agents.
        :param active_masks: (np.ndarray) denotes whether an agent is active or dead in the env.
        :param available_actions: (np.ndarray) actions available to each agent. If None, all actions are available.
        """
        if insert_mode == 'multi':
            self.share_obs[self.step + 1] = share_obs.copy()
            self.obs[self.step + 1] = obs.copy()
            self.rnn_states[self.step + 1] = rnn_states_actor.copy()
            self.rnn_states_critic[self.step + 1] = rnn_states_critic.copy()
            if self.args.use_q or self.args.sp_use_q:
                self.value_curr[self.step] = value_curr.copy()
                self.baseline[self.step] = baseline.copy()
                self.target_rnn_states_critic[self.step + 1] = target_rnn_states_critic.copy()
                self.old_probs_all[self.step] = old_probs_all.copy()
                self.old_values_all[self.step] = old_values_all.copy()
            if self.args.sp_clip:
                self.sp_value_all[self.step] = sp_value.copy()
                self.sp_prob_all[self.step] = sp_prob.copy()
            if self.args.penalty_method and not self.args.env_name == 'mujoco':
                if not self.args.sp_use_q:
                    self.penalty_old_probs_all[self.step] = penalty_old_probs.copy()

            self.actions[self.step] = actions.copy()
            self.action_log_probs[self.step] = action_log_probs.copy()
            self.value_preds[self.step] = value_preds.copy()
            self.rewards[self.step] = rewards.copy()
            self.masks[self.step + 1] = masks.copy()
            if bad_masks is not None:
                self.bad_masks[self.step + 1] = bad_masks.copy()
            if active_masks is not None:
                self.active_masks[self.step + 1] = active_masks.copy()
            if available_actions is not None and self.available_actions is not None:
                self.available_actions[self.step + 1] = available_actions.copy()
        elif insert_mode =='single':
            self.share_obs[self.step + 1,curr_rollout] = share_obs.copy()
            self.obs[self.step + 1,curr_rollout] = obs.copy()
            self.rnn_states[self.step + 1,curr_rollout] = rnn_states_actor.copy()
            self.rnn_states_critic[self.step + 1,curr_rollout] = rnn_states_critic.copy()
            if self.args.use_q or self.args.sp_use_q:
                self.value_curr[self.step,curr_rollout] = value_curr.copy()
                self.baseline[self.step, curr_rollout] = baseline.copy()
                self.old_probs_all[self.step, curr_rollout] = old_probs_all.copy()
                self.old_values_all[self.step, curr_rollout] = old_values_all.copy()
                self.target_rnn_states_critic[self.step + 1,curr_rollout] = target_rnn_states_critic.copy()
            if self.args.sp_clip:
                self.sp_value_all[self.step,curr_rollout] = sp_value.copy()
                self.sp_prob_all[self.step,curr_rollout] = sp_prob.copy()
            if self.args.penalty_method and not self.args.env_name == 'mujoco':
                if not self.args.sp_use_q:
                    self.penalty_old_probs_all[self.step,curr_rollout] = penalty_old_probs.copy()

            self.actions[self.step,curr_rollout] = actions.copy()
            self.action_log_probs[self.step,curr_rollout] = action_log_probs.copy()
            self.value_preds[self.step,curr_rollout] = value_preds.copy()
            self.rewards[self.step,curr_rollout] = rewards.copy()
            self.masks[self.step + 1,curr_rollout] = masks.copy()
            if bad_masks is not None:
                self.bad_masks[self.step + 1,curr_rollout] = bad_masks.copy()
            if active_masks is not None:
                self.active_masks[self.step + 1,curr_rollout] = active_masks.copy()
            if available_actions is not None and self.available_actions is not None:
                self.available_actions[self.step + 1,curr_rollout] = available_actions.copy()

        self.step = (self.step + 1) % self.episode_length

    def chooseinsert(self, share_obs, obs, rnn_states, rnn_states_critic, actions, action_log_probs,
                     value_preds, rewards, masks, bad_masks=None, active_masks=None, available_actions=None):
        """
        Insert data into the buffer. This insert function is used specifically for Hanabi, which is turn based.
        :param share_obs: (argparse.Namespace) arguments containing relevant model, policy, and env information.
        :param obs: (np.ndarray) local agent observations.
        :param rnn_states_actor: (np.ndarray) RNN states for actor network.
        :param rnn_states_critic: (np.ndarray) RNN states for critic network.
        :param actions:(np.ndarray) actions taken by agents.
        :param action_log_probs:(np.ndarray) log probs of actions taken by agents
        :param value_preds: (np.ndarray) value function prediction at each step.
        :param rewards: (np.ndarray) reward collected at each step.
        :param masks: (np.ndarray) denotes whether the environment has terminated or not.
        :param bad_masks: (np.ndarray) denotes indicate whether whether true terminal state or due to episode limit
        :param active_masks: (np.ndarray) denotes whether an agent is active or dead in the env.
        :param available_actions: (np.ndarray) actions available to each agent. If None, all actions are available.
        """
        self.share_obs[self.step] = share_obs.copy()
        self.obs[self.step] = obs.copy()
        self.rnn_states[self.step + 1] = rnn_states.copy()
        self.rnn_states_critic[self.step + 1] = rnn_states_critic.copy()
        self.actions[self.step] = actions.copy()
        self.action_log_probs[self.step] = action_log_probs.copy()
        self.value_preds[self.step] = value_preds.copy()
        self.rewards[self.step] = rewards.copy()
        self.masks[self.step + 1] = masks.copy()
        if bad_masks is not None:
            self.bad_masks[self.step + 1] = bad_masks.copy()
        if active_masks is not None:
            self.active_masks[self.step] = active_masks.copy()
        if available_actions is not None  and self.available_actions is not None:
            self.available_actions[self.step] = available_actions.copy()

        self.step = (self.step + 1) % self.episode_length



    def after_update(self):
        """Copy last timestep data to first index. Called after update to model."""
        self.share_obs[0] = self.share_obs[-1].copy()
        self.obs[0] = self.obs[-1].copy()
        self.rnn_states[0] = self.rnn_states[-1].copy()
        self.rnn_states_critic[0] = self.rnn_states_critic[-1].copy()
        self.masks[0] = self.masks[-1].copy()
        self.bad_masks[0] = self.bad_masks[-1].copy()
        self.active_masks[0] = self.active_masks[-1].copy()
        if self.available_actions is not None:
            self.available_actions[0] = self.available_actions[-1].copy()

        if self.args.aga_tag:
            self.update_update_tag()

    def chooseafter_update(self):
        """Copy last timestep data to first index. This method is used for Hanabi."""
        self.rnn_states[0] = self.rnn_states[-1].copy()
        self.rnn_states_critic[0] = self.rnn_states_critic[-1].copy()
        self.masks[0] = self.masks[-1].copy()
        self.bad_masks[0] = self.bad_masks[-1].copy()

    def compute_returns(self, next_value, value_normalizer=None,curr_value = None):
        """
        Compute returns either as discounted sum of rewards, or using GAE.
        :param next_value: (np.ndarray) value predictions for the step after the last episode step.
        :param value_normalizer: (PopArt) If not None, PopArt value normalizer instance.
        """
        if self._use_proper_time_limits:
            if self._use_gae:
                self.value_preds[-1] = next_value
                gae = 0
                for step in reversed(range(self.rewards.shape[0])):
                    if self._use_popart:
                        # step + 1
                        delta = self.rewards[step] + self.gamma * value_normalizer.denormalize(
                            self.value_preds[step + 1]) * self.masks[step + 1] \
                                - value_normalizer.denormalize(self.value_preds[step])
                        gae = delta + self.gamma * self.gae_lambda * gae * self.masks[step + 1]
                        gae = gae * self.bad_masks[step + 1]
                        self.returns[step] = gae + value_normalizer.denormalize(self.value_preds[step])
                    else:
                        delta = self.rewards[step] + self.gamma * self.value_preds[step + 1] * self.masks[step + 1] - \
                                self.value_preds[step]
                        gae = delta + self.gamma * self.gae_lambda * self.masks[step + 1] * gae
                        gae = gae * self.bad_masks[step + 1]
                        self.returns[step] = gae + self.value_preds[step]
            else:
                self.returns[-1] = next_value
                for step in reversed(range(self.rewards.shape[0])):
                    if self._use_popart:
                        self.returns[step] = (self.returns[step + 1] * self.gamma * self.masks[step + 1] + self.rewards[
                            step]) * self.bad_masks[step + 1] \
                                             + (1 - self.bad_masks[step + 1]) * value_normalizer.denormalize(
                            self.value_preds[step])
                    else:
                        self.returns[step] = (self.returns[step + 1] * self.gamma * self.masks[step + 1] + self.rewards[
                            step]) * self.bad_masks[step + 1] \
                                             + (1 - self.bad_masks[step + 1]) * self.value_preds[step]
        else:
            if self._use_gae:
                self.value_preds[-1] = next_value
                gae = 0
                for step in reversed(range(self.rewards.shape[0])):
                    if self._use_popart:
                        delta = self.rewards[step] + self.gamma * value_normalizer.denormalize(
                            self.value_preds[step + 1]) * self.masks[step + 1] \
                                - value_normalizer.denormalize(self.value_preds[step])
                        gae = delta + self.gamma * self.gae_lambda * self.masks[step + 1] * gae
                        self.returns[step] = gae + value_normalizer.denormalize(self.value_preds[step])
                    else:
                        delta = self.rewards[step] + self.gamma * self.value_preds[step + 1] * self.masks[step + 1] - \
                                self.value_preds[step]
                        gae = delta + self.gamma * self.gae_lambda * self.masks[step + 1] * gae
                        self.returns[step] = gae + self.value_preds[step]
            else:
                self.returns[-1] = next_value
                for step in reversed(range(self.rewards.shape[0])):
                    self.returns[step] = self.returns[step + 1] * self.gamma * self.masks[step + 1] + self.rewards[step]
        if self.args.use_q or self.args.sp_use_q:
            if self._use_proper_time_limits:
                if self._use_gae:
                    self.value_curr[-1] = next_value
            else:
                if self._use_gae:
                    # print('self.value_curr = {} curr_value = {}'.format(self.value_curr.shape,curr_value.shape))
                    self.value_curr[-1] = curr_value


    def feed_forward_generator(self, advantages, num_mini_batch=None, mini_batch_size=None):
        """
        Yield training data for MLP policies.
        :param advantages: (np.ndarray) adv
        antage estimates.
        :param num_mini_batch: (int) number of minibatches to split the batch into.
        :param mini_batch_size: (int) number of samples in each minibatch.
        """
        episode_length, n_rollout_threads, num_agents = self.rewards.shape[0:3]

        if  self.args.aga_tag and self.aga_update_tag:
            batch_size = n_rollout_threads * episode_length
        else:
            batch_size = n_rollout_threads * episode_length * num_agents

        if mini_batch_size is None:
            assert batch_size >= num_mini_batch, (
                "PPO requires the number of processes ({}) "
                "* number of steps ({}) * number of agents ({}) = {} "
                "to be greater than or equal to the number of PPO mini batches ({})."
                "".format(n_rollout_threads, episode_length, num_agents,
                          n_rollout_threads * episode_length * num_agents,
                          num_mini_batch))
            mini_batch_size = batch_size // num_mini_batch

        rand = torch.randperm(batch_size).numpy()
        sampler = [rand[i * mini_batch_size:(i + 1) * mini_batch_size] for i in range(num_mini_batch)]

        if  self.args.aga_tag and self.aga_update_tag:
            share_obs = self.share_obs[:-1,:,self.update_index].reshape(-1, *self.share_obs.shape[3:])
            obs = self.obs[:-1,:,self.update_index].reshape(-1, *self.obs.shape[3:])
            rnn_states = self.rnn_states[:-1,:,self.update_index].reshape(-1, *self.rnn_states.shape[3:])
            rnn_states_critic = self.rnn_states_critic[:-1,:,self.update_index].reshape(-1, *self.rnn_states_critic.shape[3:])
            actions = self.actions[:,:,self.update_index].reshape(-1, self.actions.shape[-1])
            if self.available_actions is not None:
                available_actions = self.available_actions[:-1,:,self.update_index].reshape(-1, self.available_actions.shape[-1])
            value_preds = self.value_preds[:-1,:,self.update_index].reshape(-1, 1)
            if self.args.use_q or self.args.sp_use_q:
                target_rnn_states_critic = self.target_rnn_states_critic[:-1, :, self.update_index].reshape(-1,
                                                                                              *self.target_rnn_states_critic.shape[
                                                                                               3:])
                value_curr = self.value_curr[:-1, :, self.update_index].reshape(-1, 1)
                curr_returns = self.curr_returns[:-1,:,self.update_index].reshape(-1, 1)
                baseline = self.baseline[:-1, :, self.update_index].reshape(-1, 1)
                old_probs_all = self.old_probs_all[:-1, :, self.update_index].reshape(-1,self.args.n_actions)
                old_values_all = self.old_values_all[:-1, :, self.update_index].reshape(-1, self.args.n_actions)
            if self.args.sp_clip:
                sp_prob_all = self.sp_prob_all[:-1,:, self.update_index].reshape(-1, self.args.sp_num)
                sp_value_all = self.sp_value_all[:-1,:, self.update_index].reshape(-1, self.args.sp_num)
            if self.args.penalty_method and not self.args.env_name == 'mujoco':
                if not self.args.sp_use_q:
                    penalty_old_probs_all = self.penalty_old_probs_all[:-1,:, self.update_index].reshape(-1, self.args.n_actions)


            returns = self.returns[:-1,:,self.update_index].reshape(-1, 1)
            masks = self.masks[:-1,:,self.update_index].reshape(-1, 1)
            active_masks = self.active_masks[:-1,:,self.update_index].reshape(-1, 1)
            action_log_probs = self.action_log_probs[:,:,self.update_index].reshape(-1, self.action_log_probs.shape[-1])
            advantages = advantages[:,:,self.update_index].reshape(-1, 1)
        else:
            share_obs = self.share_obs[:-1].reshape(-1, *self.share_obs.shape[3:])
            obs = self.obs[:-1].reshape(-1, *self.obs.shape[3:])
            rnn_states = self.rnn_states[:-1].reshape(-1, *self.rnn_states.shape[3:])
            rnn_states_critic = self.rnn_states_critic[:-1].reshape(-1, *self.rnn_states_critic.shape[3:])

            actions = self.actions.reshape(-1, self.actions.shape[-1])
            if self.available_actions is not None:
                available_actions = self.available_actions[:-1].reshape(-1, self.available_actions.shape[-1])
            value_preds = self.value_preds[:-1].reshape(-1, 1)
            if self.args.use_q or self.args.sp_use_q:
                target_rnn_states_critic = self.target_rnn_states_critic[:-1].reshape(-1,
                                                                                      *self.target_rnn_states_critic.shape[
                                                                                       3:])
                value_curr = self.value_curr[:-1].reshape(-1, 1)
                curr_returns = self.curr_returns[:-1].reshape(-1, 1)
                baseline = self.baseline[:-1].reshape(-1,1)
                old_probs_all = self.old_probs_all[:-1].reshape(-1, self.args.n_actions)
                old_values_all = self.old_values_all[:-1].reshape(-1, self.args.n_actions)
            if self.args.sp_clip:
                sp_prob_all = self.sp_prob_all[:-1].reshape(-1, self.args.sp_num)
                sp_value_all = self.sp_value_all[:-1].reshape(-1, self.args.sp_num)
            if self.args.penalty_method and not self.args.env_name == 'mujoco':
                if not self.args.sp_use_q:
                    penalty_old_probs_all = self.penalty_old_probs_all[:-1].reshape(-1,self.args.n_actions)

            returns = self.returns[:-1].reshape(-1, 1)
            masks = self.masks[:-1].reshape(-1, 1)
            active_masks = self.active_masks[:-1].reshape(-1, 1)
            action_log_probs = self.action_log_probs.reshape(-1, self.action_log_probs.shape[-1])
            advantages = advantages.reshape(-1, 1)

        for indices in sampler:
            # obs size [T+1 N M Dim]-->[T N M Dim]-->[T*N*M,Dim]-->[index,Dim]
            share_obs_batch = share_obs[indices]
            obs_batch = obs[indices]
            rnn_states_batch = rnn_states[indices]
            rnn_states_critic_batch = rnn_states_critic[indices]
            if self.args.use_q or self.args.sp_use_q:
                target_rnn_states_critic_batch = target_rnn_states_critic[indices]
                value_curr_batch = value_curr[indices]
                baseline_batch = baseline[indices]
                old_probs_batch = old_probs_all[indices]
                old_values_batch = old_values_all[indices]
                curr_returns_batch = curr_returns[indices]
            if self.args.sp_clip:
                sp_prob_batch = sp_prob_all[indices]
                sp_value_batch = sp_value_all[indices]
            if self.args.penalty_method and not self.args.env_name == 'mujoco':
                if not self.args.sp_use_q:
                    penalty_old_probs_batch = penalty_old_probs_all[indices]
                else:
                    penalty_old_probs_batch = None
            actions_batch = actions[indices]
            if self.available_actions is not None:
                available_actions_batch = available_actions[indices]
            else:
                available_actions_batch = None
            value_preds_batch = value_preds[indices]
            return_batch = returns[indices]

            masks_batch = masks[indices]
            active_masks_batch = active_masks[indices]
            old_action_log_probs_batch = action_log_probs[indices]
            if advantages is None:
                adv_targ = None
            else:
                adv_targ = advantages[indices]

            base_ret = [share_obs_batch, obs_batch, rnn_states_batch, rnn_states_critic_batch, actions_batch,\
                          value_preds_batch, return_batch, masks_batch, active_masks_batch, old_action_log_probs_batch,\
                          adv_targ, available_actions_batch]
            if self.args.sp_clip:
                sp_ret = [sp_value_batch,sp_prob_batch]
            else:
                sp_ret = None
            if self.args.use_q or self.args.sp_use_q:
                q_ret = [target_rnn_states_critic_batch,value_curr_batch,\
                          curr_returns_batch,baseline_batch,old_probs_batch,old_values_batch]
            else:
                q_ret = None

            if self.args.penalty_method and not self.args.env_name == 'mujoco':
                if not self.args.sp_use_q:
                    penalty_ret = [penalty_old_probs_batch]
                else:
                    penalty_ret = None
            else:
                penalty_ret = None


            yield base_ret,q_ret,sp_ret,penalty_ret


    def update_update_tag(self):
        if self.args.aga_tag and self.aga_update_tag:
            self.update_index = (self.update_index + 1) % self.n_agents
        self.total_update_step += 1
        if self.args.aga_first:
            self.aga_update_tag =  (self.total_update_step % self.total_period) <= (self.total_period - self.iql_period)
            # if self.args.target_dec and not self.args.soft_target:
            #     if (self.total_update_step % self.total_period) == (self.total_period - self.iql_period):
            #         self.hard_update_critic()
        else:
            self.aga_update_tag =  (self.total_update_step % self.total_period) >= self.iql_period
            # if self.args.target_dec and not self.args.soft_target:
            #     if (self.total_update_step % self.total_period) == 0:
            #         self.hard_update_critic()
        # if self.args.target_dec and self.args.soft_target:
        #     self.soft_update_critic()

    def naive_recurrent_generator(self, advantages, num_mini_batch):
        """
        Yield training data for non-chunked RNN training.
        :param advantages: (np.ndarray) advantage estimates.
        :param num_mini_batch: (int) number of minibatches to split the batch into.
        """
        episode_length, n_rollout_threads, num_agents = self.rewards.shape[0:3]
        batch_size = n_rollout_threads * num_agents
        assert n_rollout_threads * num_agents >= num_mini_batch, (
            "PPO requires the number of processes ({})* number of agents ({}) "
            "to be greater than or equal to the number of "
            "PPO mini batches ({}).".format(n_rollout_threads, num_agents, num_mini_batch))
        num_envs_per_batch = batch_size // num_mini_batch
        perm = torch.randperm(batch_size).numpy()

        share_obs = self.share_obs.reshape(-1, batch_size, *self.share_obs.shape[3:])
        obs = self.obs.reshape(-1, batch_size, *self.obs.shape[3:])
        rnn_states = self.rnn_states.reshape(-1, batch_size, *self.rnn_states.shape[3:])
        rnn_states_critic = self.rnn_states_critic.reshape(-1, batch_size, *self.rnn_states_critic.shape[3:])
        actions = self.actions.reshape(-1, batch_size, self.actions.shape[-1])
        if self.available_actions is not None:
            available_actions = self.available_actions.reshape(-1, batch_size, self.available_actions.shape[-1])
        value_preds = self.value_preds.reshape(-1, batch_size, 1)
        returns = self.returns.reshape(-1, batch_size, 1)
        masks = self.masks.reshape(-1, batch_size, 1)
        active_masks = self.active_masks.reshape(-1, batch_size, 1)
        action_log_probs = self.action_log_probs.reshape(-1, batch_size, self.action_log_probs.shape[-1])
        advantages = advantages.reshape(-1, batch_size, 1)

        for start_ind in range(0, batch_size, num_envs_per_batch):
            share_obs_batch = []
            obs_batch = []
            rnn_states_batch = []
            rnn_states_critic_batch = []
            actions_batch = []
            available_actions_batch = []
            value_preds_batch = []
            return_batch = []
            masks_batch = []
            active_masks_batch = []
            old_action_log_probs_batch = []
            adv_targ = []

            for offset in range(num_envs_per_batch):
                ind = perm[start_ind + offset]
                share_obs_batch.append(share_obs[:-1, ind])
                obs_batch.append(obs[:-1, ind])
                rnn_states_batch.append(rnn_states[0:1, ind])
                rnn_states_critic_batch.append(rnn_states_critic[0:1, ind])
                actions_batch.append(actions[:, ind])
                if self.available_actions is not None:
                    available_actions_batch.append(available_actions[:-1, ind])
                value_preds_batch.append(value_preds[:-1, ind])
                return_batch.append(returns[:-1, ind])
                masks_batch.append(masks[:-1, ind])
                active_masks_batch.append(active_masks[:-1, ind])
                old_action_log_probs_batch.append(action_log_probs[:, ind])
                adv_targ.append(advantages[:, ind])

            # [N[T, dim]]
            T, N = self.episode_length, num_envs_per_batch
            # These are all from_numpys of size (T, N, -1)
            share_obs_batch = np.stack(share_obs_batch, 1)
            obs_batch = np.stack(obs_batch, 1)
            actions_batch = np.stack(actions_batch, 1)
            if self.available_actions is not None:
                available_actions_batch = np.stack(available_actions_batch, 1)
            value_preds_batch = np.stack(value_preds_batch, 1)
            return_batch = np.stack(return_batch, 1)
            masks_batch = np.stack(masks_batch, 1)
            active_masks_batch = np.stack(active_masks_batch, 1)
            old_action_log_probs_batch = np.stack(old_action_log_probs_batch, 1)
            adv_targ = np.stack(adv_targ, 1)

            # States is just a (N, dim) from_numpy [N[1,dim]]
            rnn_states_batch = np.stack(rnn_states_batch).reshape(N, *self.rnn_states.shape[3:])
            rnn_states_critic_batch = np.stack(rnn_states_critic_batch).reshape(N, *self.rnn_states_critic.shape[3:])

            # Flatten the (T, N, ...) from_numpys to (T * N, ...)
            share_obs_batch = _flatten(T, N, share_obs_batch)
            obs_batch = _flatten(T, N, obs_batch)
            actions_batch = _flatten(T, N, actions_batch)
            if self.available_actions is not None:
                available_actions_batch = _flatten(T, N, available_actions_batch)
            else:
                available_actions_batch = None
            value_preds_batch = _flatten(T, N, value_preds_batch)
            return_batch = _flatten(T, N, return_batch)
            masks_batch = _flatten(T, N, masks_batch)
            active_masks_batch = _flatten(T, N, active_masks_batch)
            old_action_log_probs_batch = _flatten(T, N, old_action_log_probs_batch)
            adv_targ = _flatten(T, N, adv_targ)

            yield share_obs_batch, obs_batch, rnn_states_batch, rnn_states_critic_batch, actions_batch,\
                  value_preds_batch, return_batch, masks_batch, active_masks_batch, old_action_log_probs_batch,\
                  adv_targ, available_actions_batch

    def recurrent_generator(self, advantages, num_mini_batch, data_chunk_length):
        """
        Yield training data for chunked RNN training.
        :param advantages: (np.ndarray) advantage estimates.
        :param num_mini_batch: (int) number of minibatches to split the batch into.
        :param data_chunk_length: (int) length of sequence chunks with which to train RNN.
        """
        episode_length, n_rollout_threads, num_agents = self.rewards.shape[0:3]
        batch_size = n_rollout_threads * episode_length * num_agents
        data_chunks = batch_size // data_chunk_length  # [C=r*T*M/L]
        mini_batch_size = data_chunks // num_mini_batch

        rand = torch.randperm(data_chunks).numpy()
        sampler = [rand[i * mini_batch_size:(i + 1) * mini_batch_size] for i in range(num_mini_batch)]

        if len(self.share_obs.shape) > 4:
            share_obs = self.share_obs[:-1].transpose(1, 2, 0, 3, 4, 5).reshape(-1, *self.share_obs.shape[3:])
            obs = self.obs[:-1].transpose(1, 2, 0, 3, 4, 5).reshape(-1, *self.obs.shape[3:])
        else:
            share_obs = _cast(self.share_obs[:-1])
            obs = _cast(self.obs[:-1])

        actions = _cast(self.actions)
        action_log_probs = _cast(self.action_log_probs)
        advantages = _cast(advantages)
        value_preds = _cast(self.value_preds[:-1])
        returns = _cast(self.returns[:-1])
        masks = _cast(self.masks[:-1])
        active_masks = _cast(self.active_masks[:-1])
        # rnn_states = _cast(self.rnn_states[:-1])
        # rnn_states_critic = _cast(self.rnn_states_critic[:-1])
        rnn_states = self.rnn_states[:-1].transpose(1, 2, 0, 3, 4).reshape(-1, *self.rnn_states.shape[3:])
        rnn_states_critic = self.rnn_states_critic[:-1].transpose(1, 2, 0, 3, 4).reshape(-1,
                                                                                         *self.rnn_states_critic.shape[
                                                                                          3:])

        if self.available_actions is not None:
            available_actions = _cast(self.available_actions[:-1])

        for indices in sampler:
            share_obs_batch = []
            obs_batch = []
            rnn_states_batch = []
            rnn_states_critic_batch = []
            actions_batch = []
            available_actions_batch = []
            value_preds_batch = []
            return_batch = []
            masks_batch = []
            active_masks_batch = []
            old_action_log_probs_batch = []
            adv_targ = []

            for index in indices:

                ind = index * data_chunk_length
                # size [T+1 N M Dim]-->[T N M Dim]-->[N,M,T,Dim]-->[N*M*T,Dim]-->[L,Dim]
                share_obs_batch.append(share_obs[ind:ind + data_chunk_length])
                obs_batch.append(obs[ind:ind + data_chunk_length])
                actions_batch.append(actions[ind:ind + data_chunk_length])
                if self.available_actions is not None:
                    available_actions_batch.append(available_actions[ind:ind + data_chunk_length])
                value_preds_batch.append(value_preds[ind:ind + data_chunk_length])
                return_batch.append(returns[ind:ind + data_chunk_length])
                masks_batch.append(masks[ind:ind + data_chunk_length])
                active_masks_batch.append(active_masks[ind:ind + data_chunk_length])
                old_action_log_probs_batch.append(action_log_probs[ind:ind + data_chunk_length])
                adv_targ.append(advantages[ind:ind + data_chunk_length])
                # size [T+1 N M Dim]-->[T N M Dim]-->[N M T Dim]-->[N*M*T,Dim]-->[1,Dim]
                rnn_states_batch.append(rnn_states[ind])
                rnn_states_critic_batch.append(rnn_states_critic[ind])

            L, N = data_chunk_length, mini_batch_size

            # These are all from_numpys of size (L, N, Dim)           
            share_obs_batch = np.stack(share_obs_batch, axis=1)
            obs_batch = np.stack(obs_batch, axis=1)

            actions_batch = np.stack(actions_batch, axis=1)
            if self.available_actions is not None:
                available_actions_batch = np.stack(available_actions_batch, axis=1)
            value_preds_batch = np.stack(value_preds_batch, axis=1)
            return_batch = np.stack(return_batch, axis=1)
            masks_batch = np.stack(masks_batch, axis=1)
            active_masks_batch = np.stack(active_masks_batch, axis=1)
            old_action_log_probs_batch = np.stack(old_action_log_probs_batch, axis=1)
            adv_targ = np.stack(adv_targ, axis=1)

            # States is just a (N, -1) from_numpy
            rnn_states_batch = np.stack(rnn_states_batch).reshape(N, *self.rnn_states.shape[3:])
            rnn_states_critic_batch = np.stack(rnn_states_critic_batch).reshape(N, *self.rnn_states_critic.shape[3:])

            # Flatten the (L, N, ...) from_numpys to (L * N, ...)
            share_obs_batch = _flatten(L, N, share_obs_batch)
            obs_batch = _flatten(L, N, obs_batch)
            actions_batch = _flatten(L, N, actions_batch)
            if self.available_actions is not None:
                available_actions_batch = _flatten(L, N, available_actions_batch)
            else:
                available_actions_batch = None
            value_preds_batch = _flatten(L, N, value_preds_batch)
            return_batch = _flatten(L, N, return_batch)
            masks_batch = _flatten(L, N, masks_batch)
            active_masks_batch = _flatten(L, N, active_masks_batch)
            old_action_log_probs_batch = _flatten(L, N, old_action_log_probs_batch)
            adv_targ = _flatten(L, N, adv_targ)

            yield share_obs_batch, obs_batch, rnn_states_batch, rnn_states_critic_batch, actions_batch,\
                  value_preds_batch, return_batch, masks_batch, active_masks_batch, old_action_log_probs_batch,\
                  adv_targ, available_actions_batch
