from collections import OrderedDict

import torch
import torch.nn as nn
import torch.nn.functional as F
import torch.distributed as dist
from torch.distributions import Bernoulli
from torchmeta.modules import (MetaModule, MetaSequential, MetaConv2d,
                               MetaBatchNorm2d, MetaLinear)

"""
ResNet
from https://github.com/kjunelee/MetaOptNet
This ResNet network was designed following the practice of the following papers:
TADAM: Task dependent adaptive metric for improved few-shot learning (Oreshkin et al., in NIPS 2018) and
A Simple Neural Attentive Meta-Learner (Mishra et al., in ICLR 2018).
"""


def get_subdict(adict, name):
    if adict is None:
        return adict
    tmp = {k[len(name) + 1:]:adict[k] for k in adict if name in k}
    return tmp

class DropBlock(MetaModule):
    def __init__(self, block_size):
        super(DropBlock, self).__init__()

        self.block_size = block_size
        # self.gamma = gamma
        # self.bernouli = Bernoulli(gamma)

    def forward(self, x, gamma):
        # shape: (bsize, channels, height, width)

        if self.training:
            batch_size, channels, height, width = x.shape

            bernoulli = Bernoulli(gamma)
            mask = bernoulli.sample(
                (batch_size, channels, height - (self.block_size - 1), width - (self.block_size - 1))).cuda()
            # print((x.sample[-2], x.sample[-1]))
            block_mask = self._compute_block_mask(mask)
            # print (block_mask.size())
            # print (x.size())
            countM = block_mask.size()[0] * block_mask.size()[1] * block_mask.size()[2] * block_mask.size()[3]
            count_ones = block_mask.sum()

            return block_mask * x * (countM / count_ones)
        else:
            return x

    def _compute_block_mask(self, mask):
        left_padding = int((self.block_size - 1) / 2)
        right_padding = int(self.block_size / 2)

        batch_size, channels, height, width = mask.shape
        # print ("mask", mask[0][0])
        non_zero_idxs = mask.nonzero()
        nr_blocks = non_zero_idxs.shape[0]

        offsets = torch.stack(
            [
                torch.arange(self.block_size).view(-1, 1).expand(self.block_size, self.block_size).reshape(-1),
                # - left_padding,
                torch.arange(self.block_size).repeat(self.block_size),  # - left_padding
            ]
        ).t().cuda()
        offsets = torch.cat((torch.zeros(self.block_size ** 2, 2).cuda().long(), offsets.long()), 1)

        if nr_blocks > 0:
            non_zero_idxs = non_zero_idxs.repeat(self.block_size ** 2, 1)
            offsets = offsets.repeat(nr_blocks, 1).view(-1, 4)
            offsets = offsets.long()

            block_idxs = non_zero_idxs + offsets
            # block_idxs += left_padding
            padded_mask = F.pad(mask, (left_padding, right_padding, left_padding, right_padding))
            padded_mask[block_idxs[:, 0], block_idxs[:, 1], block_idxs[:, 2], block_idxs[:, 3]] = 1.
        else:
            padded_mask = F.pad(mask, (left_padding, right_padding, left_padding, right_padding))

        block_mask = 1 - padded_mask  # [:height, :width]
        return block_mask


class BasicBlock(MetaModule):
    expansion = 1

    def __init__(self, inplanes, planes, stride=1,
                downsample=None, drop_block=False,
                block_size=1, max_padding=0, track_running_stats=False):
        super(BasicBlock, self).__init__()
        self.conv1 = MetaConv2d(inplanes, planes, kernel_size=3,
                                stride=1, padding=1, bias=False)
        self.bn1 = MetaBatchNorm2d(planes, track_running_stats=track_running_stats)
        self.relu1 = nn.LeakyReLU()
        self.conv2 = MetaConv2d(planes, planes, kernel_size=3,
                                stride=1, padding=1, bias=False)
        self.bn2 = MetaBatchNorm2d(planes, track_running_stats=track_running_stats)
        self.relu2 = nn.LeakyReLU()
        self.conv3 = MetaConv2d(planes, planes, kernel_size=3,
                                stride=1, padding=1, bias=False)
        self.bn3 = MetaBatchNorm2d(planes, track_running_stats=track_running_stats)
        self.relu3 = nn.LeakyReLU()

        self.maxpool = nn.MaxPool2d(stride=stride, kernel_size=[stride,stride],
                                                            padding=max_padding)
        self.max_pool = True if stride != max_padding else False
        self.downsample = downsample
        self.stride = stride
        self.num_batches_tracked = 0
        self.drop_block = drop_block
        self.block_size = block_size
        self.DropBlock = DropBlock(block_size=self.block_size)

    def forward(self, x, params=None):
        self.num_batches_tracked += 1

        residual = x
        out = self.conv1(x, params=get_subdict(params, 'conv1'))
        out = self.bn1(out, params=get_subdict(params, 'bn1'))
        out = self.relu1(out)

        out = self.conv2(out, params=get_subdict(params, 'conv2'))
        out = self.bn2(out, params=get_subdict(params, 'bn2'))
        out = self.relu2(out)

        out = self.conv3(out, params=get_subdict(params, 'conv3'))
        out = self.bn3(out, params=get_subdict(params, 'bn3'))

        if self.downsample is not None:
            residual = self.downsample(x, params=get_subdict(params, 'downsample'))
        out += residual
        out = self.relu3(out)

        if self.max_pool:
            out = self.maxpool(out)

        return out


def mlp(in_dim, out_dim):
    return MetaSequential(OrderedDict([
        ('linear1', MetaLinear(in_dim, out_dim, bias=True)),
        ('relu', nn.ReLU()),
        ('linear2', MetaLinear(out_dim, out_dim, bias=True)),
    ]))


def mlp_drop(in_dim, out_dim):
    return MetaSequential(OrderedDict([
        ('linear1', MetaLinear(in_dim, out_dim, bias=True)),
        ('relu', nn.ReLU()),
        ('linear2', MetaLinear(out_dim, out_dim, bias=True)),
    ]))


class ResNet(MetaModule):
    def __init__(self, blocks, avg_pool=True, dropblock_size=5,
                 out_features=5, wh_size=1, inductive_bn=False):
        self.inplanes = 3
        super(ResNet, self).__init__()
        self.anil = False

        self.inductive_bn = inductive_bn
        self.layer1 = self._make_layer(blocks[0], 64, stride=2, drop_block=True,
                                       block_size=dropblock_size)
        self.layer2 = self._make_layer(blocks[1], 128, stride=2, drop_block=True,
                                       block_size=dropblock_size)
        self.layer3 = self._make_layer(blocks[2], 256, stride=2, drop_block=True,
                                       block_size=dropblock_size)
        self.layer4 = self._make_layer(blocks[3], 512, stride=2, drop_block=True,
                                       block_size=dropblock_size)

        if avg_pool:
            self.avgpool = nn.AdaptiveAvgPool2d(1)
        
        self.keep_avg_pool = avg_pool
        self.classifier = MetaLinear(512 * wh_size * wh_size, out_features)

        for m in self.modules():
            if isinstance(m, MetaConv2d):
                nn.init.xavier_uniform_(m.weight)
            if isinstance(m, MetaLinear):
                nn.init.xavier_uniform_(m.weight)

    def _make_layer(self, block, planes, stride=1,
                    drop_block=False, block_size=1, max_padding=0):
        downsample = None
        if stride != 1 or self.inplanes != planes * block.expansion:
            downsample = MetaSequential(
                MetaConv2d(self.inplanes, planes * block.expansion,
                                kernel_size=1, stride=1, bias=False),
                MetaBatchNorm2d(planes * block.expansion,
                                track_running_stats=self.inductive_bn),
            )

        layers = []
        layers.append(block(self.inplanes, planes, stride,
                    downsample, drop_block, block_size, max_padding, track_running_stats=self.inductive_bn))
        self.inplanes = planes * block.expansion
        return MetaSequential(*layers)
    
    def _forward_all(self, x, params, inner_update_type):
        if self.anil or inner_update_type=='linear_only':
            params_feature = [None for _ in range(4)]
        else:
            params_feature = [get_subdict(params, f'layer{i+1}') for i in range(4)]

        x = self.layer1(x, params=params_feature[0])
        x = self.layer2(x, params=params_feature[1])
        x = self.layer3(x, params=params_feature[2])
        x = self.layer4(x, params=params_feature[3])
        if self.keep_avg_pool:
            x = self.avgpool(x)
        features = x.view((x.size(0), -1))
        
        if inner_update_type == 'encoder_only':
            logits = self.classifier(features)
        else:
            logits = self.classifier(features, params=get_subdict(params, 'classifier'))
       
        return logits, features

    def forward(self, qry, adv=None, sprt=None, qry_num=1, adv_num=0, sprt_num=0, params=None, params2=None, feat=False, inner_update_type='both'):
        

        if qry_num == 1:
            x1 = qry
        else:
            x1, x2 = qry
        
        logits_qry, z_qry = self._forward_all(x1, params, inner_update_type)

        if qry_num == 2:
            logits_qry2, z_qry2 = self._forward_all(x2, params2, inner_update_type)
            logits_qry = (logits_qry, logits_qry2)
            z_qry = (z_qry, z_qry2)

        if adv_num == 1:
            adv1 = adv
        elif adv_num == 2:
            adv1, adv2 = adv
        
        if adv_num >= 1:
            logits_adv, z_adv = self._forward_all(adv1, params, inner_update_type)
        else:
            logits_adv, z_adv = None, None
        if adv_num == 2:
            logits_adv2, z_adv2 = self._forward_all(adv2, params2, inner_update_type)
            logits_adv = (logits_adv, logits_adv2)
            z_adv = (z_adv, z_adv2)

        if sprt_num == 1:
            sprt1 = sprt
        elif sprt_num == 2:
            sprt1, sprt2 = sprt

        if sprt_num >= 1:
            logits_sprt, z_sprt = self._forward_all(sprt1, params, inner_update_type)
        else:
            logits_sprt, z_sprt = None, None
        if sprt_num == 2:
            logits_sprt2, z_sprt2 = self._forward_all(sprt2, params2, inner_update_type)
            logits_sprt = (logits_sprt, logits_sprt2)
            z_sprt = (z_sprt, z_sprt2)

        if feat:
            if adv_num>0 or sprt_num>0:
                return logits_qry, logits_adv, logits_sprt, z_qry, z_adv, z_sprt
            else:
                return logits_qry, z_qry
        else:
            if adv_num>0 or sprt_num>0:
                return logits_qry, logits_adv, logits_sprt
            else:
                return logits_qry
    
    

def ResNet12_selfsup(num_classes, inductive_bn=False):
    blocks = [BasicBlock, BasicBlock, BasicBlock, BasicBlock]
    return ResNet(blocks, out_features=num_classes, inductive_bn=inductive_bn)
