import torch

from torch import nn
from torch.nn import functional as F
import math
from .base_networks import wrn
# from .base import mxresnet
import torchvision.models as models


def get_model(model_name, train_set=None):
    if model_name in ["linear", "logistic"]:
        batch = train_set[0]
        model = Mlp_model(input_size=batch['images'].shape[0], hidden_sizes=[], n_classes=2, bias=False)
        

    if model_name == "mlp":
        model = Mlp_model()

    if model_name == "wrn":
        model = wrn.WideResNet(depth=28, num_classes=200)

    if model_name == "rnn_nlp":
        model = rnn_nlp.RNNModel(ntoken=len(train_set.corpus.dictionary), tie_weights=False)
         
    if model_name == "mlp_dropout":
        model = Mlp(n_classes=10, dropout=True)

    elif model_name == "resnet34":
        model = ResNet([3, 4, 6, 3], num_classes=10)

    elif model_name == "resnet34_100":
        model = ResNet([3, 4, 6, 3], num_classes=100)

    elif model_name == "resnet34_200":
        model = ResNet([3, 4, 6, 3], num_classes=200)

    elif model_name == "resnet34_nobn":
        model = ResNet([3, 4, 6, 3], num_classes=10)
        model.apply(deactivate_batchnorm)

    elif model_name == "resnet34_100_nobn":
        model = ResNet([3, 4, 6, 3], num_classes=100)
        model.apply(deactivate_batchnorm)

    elif model_name == "resnet18":
        model = models.resnet18()



    elif model_name == "resnet50":
        model = models.resnet50()
    
    elif model_name == "resnet101":
        model = models.resnet101()
    
    elif model_name == "resnet152":
        model = models.resnet152()
    
    elif model_name == "mxresnet50":
        model = mxresnet.mxresnet50(sa=1)

    elif model_name == "wide_resnet101":
        model = models.wide_resnet101_2()

    elif model_name == "densenet121":
        model = DenseNet121(num_classes=10)

    elif model_name == "densenet121_100":
        model = DenseNet121(num_classes=100)

    elif model_name == "densenet121_nobn":
        model = DenseNet121(num_classes=10)
        model.apply(deactivate_batchnorm)

    elif model_name == "densenet121_100_nobn":
        model = DenseNet121(num_classes=100)
        model.apply(deactivate_batchnorm)

    elif model_name == "matrix_fac_1":
        model = LinearNetwork(6, [1], 10, bias=False)
        
    elif model_name == "matrix_fac_4":
        model = LinearNetwork(6, [4], 10, bias=False)
    
    elif model_name == "matrix_fac_10":
        model = LinearNetwork(6, [10], 10, bias=False)

    elif model_name == "linear_fac":
        model = LinearNetwork(6, [], 10, bias=False)

    elif model_name == "rnn":
        time_step = 28         # rnn time step / image height
        input_size = 784//time_step         # rnn input size / image width
        hidden_units = 64

        model = RNN(time_step=time_step, 
                    input_size=input_size, 
                    hidden_units=10)

    return model

def deactivate_batchnorm(m):
    if isinstance(m, nn.BatchNorm2d):
        m.reset_parameters()
        m.eval()
        with torch.no_grad():
            m.weight.fill_(1.0)
            m.bias.zero_()
        m.weight.requires_grad = False
        m.bias.requires_grad = False

# =====================================================
# MLP
def Mlp_model(input_size=784, hidden_sizes=[512, 256], n_classes=10, bias=True, dropout=False):
    modules = []
    if len(hidden_sizes) == 0:
        modules.append(nn.Linear(input_size, n_classes, bias=bias))
    else:
        for i, layer in enumerate(hidden_sizes):
            if i == 0:
                modules.append(nn.Linear(input_size, layer, bias=bias))
            else:
                modules.append(nn.Linear(hidden_sizes[i-1], layer, bias=bias))

            modules.append(nn.ReLU())
            if dropout:
                modules.append(nn.Dropout(p=0.5))

        modules.append(nn.Linear(hidden_sizes[-1], n_classes, bias=bias))

    return nn.Sequential(*modules)

# =====================================================
# RNN
class RNN(nn.Module):
    def __init__(self, time_step, input_size, hidden_units):
        super().__init__()

        self.rnn = nn.LSTM(         # if use nn.RNN(), it hardly learns
            input_size=input_size,
            hidden_size=hidden_units,         # rnn hidden unit
            num_layers=1,           # number of rnn layer
            batch_first=True,       # input & output will has batch size as 1s dimension. e.g. (batch, time_step, input_size)
        )

        self.out = nn.Linear(hidden_units, 10)
        # initialize based on that Hinton's paper
        self.init_weights()
        self.rnn.flatten_parameters()

    def forward(self, x):
        # x shape (batch, time_step, input_size)
        # r_out shape (batch, time_step, output_size)
        # h_n shape (n_layers, batch, hidden_size)
        # h_c shape (n_layers, batch, hidden_size)
        self.rnn.flatten_parameters()
        r_out, (h_n, h_c) = self.rnn(x, None)   # None represents zero initial hidden state

        # choose r_out at the last time step
        out = F.relu(r_out[:, -1, :])
        out = self.out(out)
        return out

    def init_weights(self):
        for m in self.modules():
            if type(m) in [nn.GRU, nn.LSTM, nn.RNN]:
                for name, param in m.named_parameters():
                    if 'weight_ih' in name:
                        torch.nn.init.normal_(param.data, std=0.001)
                    elif 'weight_hh' in name:
                        torch.nn.init.eye_(param.data)
                    elif 'bias' in name:
                        param.data.fill_(0)

# =====================================================
# Linear Network
class LinearNetwork(nn.Module):
    def __init__(self, input_size, hidden_sizes, output_size, bias=True):
        super().__init__()

        # iterate averaging:
        self._prediction_params = None

        self.input_size = input_size
        if output_size:
            self.output_size = output_size
            self.squeeze_output = False
        else :
            self.output_size = 1
            self.squeeze_output = True

        if len(hidden_sizes) == 0:
            self.hidden_layers = []
            self.output_layer = nn.Linear(self.input_size, self.output_size, bias=bias)
        else:
            self.hidden_layers = nn.ModuleList([nn.Linear(in_size, out_size, bias=bias) for in_size, out_size in zip([self.input_size] + hidden_sizes[:-1], hidden_sizes)])
            self.output_layer = nn.Linear(hidden_sizes[-1], self.output_size, bias=bias)

    def forward(self, x):
        '''
            x: The input patterns/features.
        '''
        x = x.view(-1, self.input_size)
        out = x

        for layer in self.hidden_layers:
            Z = layer(out)
            # no activation in linear network.
            out = Z

        logits = self.output_layer(out)
        if self.squeeze_output:
            logits = torch.squeeze(logits)

        return logits

# =====================================================
# Logistic
class LinearRegression(torch.nn.Module):
    def __init__(self, input_dim, output_dim):
        super().__init__()
        self.linear = torch.nn.Linear(input_dim, output_dim, bias=False)

    def forward(self, x):
        outputs = self.linear(x)
        return outputs

# =====================================================
# MLP
class Mlp(nn.Module):
    def __init__(self, input_size=784,
                 hidden_sizes=[512, 256],
                 n_classes=10,
                 bias=True, dropout=False):
        super().__init__()

        self.dropout=dropout
        self.input_size = input_size
        self.hidden_layers = nn.ModuleList([nn.Linear(in_size, out_size, bias=bias) for
                                            in_size, out_size in zip([self.input_size] + hidden_sizes[:-1], hidden_sizes)])
        self.output_layer = nn.Linear(hidden_sizes[-1], n_classes, bias=bias)

    def forward(self, x):
        x = x.view(-1, self.input_size)
        out = x
        for layer in self.hidden_layers:
            Z = layer(out)
            out = F.relu(Z)

            if self.dropout:
                out = F.dropout(out, p=0.5)

        logits = self.output_layer(out)

        return logits

# =====================================================
# ResNet
class ResNet(nn.Module):
    def __init__(self, num_blocks, num_classes=10):
        super().__init__()
        block = BasicBlock
        self.in_planes = 64

        self.conv1 = nn.Conv2d(
            3, 64, kernel_size=3, stride=1, padding=1, bias=False)
        self.bn1 = nn.BatchNorm2d(64)
        self.layer1 = self._make_layer(block, 64, num_blocks[0], stride=1)
        self.layer2 = self._make_layer(block, 128, num_blocks[1], stride=2)
        self.layer3 = self._make_layer(block, 256, num_blocks[2], stride=2)
        self.layer4 = self._make_layer(block, 512, num_blocks[3], stride=2)
        self.linear = nn.Linear(512 * block.expansion, num_classes)

    def _make_layer(self, block, planes, num_blocks, stride):
        strides = [stride] + [1] * (num_blocks - 1)
        layers = []
        for stride in strides:
            layers.append(block(self.in_planes, planes, stride))
            self.in_planes = planes * block.expansion
        return nn.Sequential(*layers)

    def forward(self, x):
        out = F.relu(self.bn1(self.conv1(x)))
        out = self.layer1(out)
        out = self.layer2(out)
        out = self.layer3(out)
        out = self.layer4(out)
        out = F.avg_pool2d(out, 4)
        out = out.view(out.size(0), -1)
        out = self.linear(out)
        return out


class BasicBlock(nn.Module):
    expansion = 1

    def __init__(self, in_planes, planes, stride=1):
        super(BasicBlock, self).__init__()
        self.conv1 = nn.Conv2d(
            in_planes,
            planes,
            kernel_size=3,
            stride=stride,
            padding=1,
            bias=False)
        self.bn1 = nn.BatchNorm2d(planes)
        self.conv2 = nn.Conv2d(
            planes, planes, kernel_size=3, stride=1, padding=1, bias=False)
        self.bn2 = nn.BatchNorm2d(planes)

        self.shortcut = nn.Sequential()
        if stride != 1 or in_planes != self.expansion * planes:
            self.shortcut = nn.Sequential(
                nn.Conv2d(
                    in_planes,
                    self.expansion * planes,
                    kernel_size=1,
                    stride=stride,
                    bias=False), nn.BatchNorm2d(self.expansion * planes))

    def forward(self, x):
        out = F.relu(self.bn1(self.conv1(x)))
        out = self.bn2(self.conv2(out))
        out += self.shortcut(x)
        out = F.relu(out)
        return out


class Bottleneck(nn.Module):
    expansion = 4

    def __init__(self, in_planes, planes, stride=1):
        super(Bottleneck, self).__init__()
        self.conv1 = nn.Conv2d(in_planes, planes, kernel_size=1, bias=False)
        self.bn1 = nn.BatchNorm2d(planes)
        self.conv2 = nn.Conv2d(
            planes,
            planes,
            kernel_size=3,
            stride=stride,
            padding=1,
            bias=False)
        self.bn2 = nn.BatchNorm2d(planes)
        self.conv3 = nn.Conv2d(
            planes, self.expansion * planes, kernel_size=1, bias=False)
        self.bn3 = nn.BatchNorm2d(self.expansion * planes)

        self.shortcut = nn.Sequential()
        if stride != 1 or in_planes != self.expansion * planes:
            self.shortcut = nn.Sequential(
                nn.Conv2d(
                    in_planes,
                    self.expansion * planes,
                    kernel_size=1,
                    stride=stride,
                    bias=False), nn.BatchNorm2d(self.expansion * planes))

    def forward(self, x):
        out = F.relu(self.bn1(self.conv1(x)))
        out = F.relu(self.bn2(self.conv2(out)))
        out = self.bn3(self.conv3(out))
        out += self.shortcut(x)
        out = F.relu(out)
        return out


class Bottleneck_DenseNet(nn.Module):
    def __init__(self, in_planes, growth_rate):
        super().__init__()
        self.bn1 = nn.BatchNorm2d(in_planes)
        self.conv1 = nn.Conv2d(in_planes, 4*growth_rate, kernel_size=1, bias=False)
        self.bn2 = nn.BatchNorm2d(4*growth_rate)
        self.conv2 = nn.Conv2d(4*growth_rate, growth_rate, kernel_size=3, padding=1, bias=False)

    def forward(self, x):
        out = self.conv1(F.relu(self.bn1(x)))
        out = self.conv2(F.relu(self.bn2(out)))
        out = torch.cat([out,x], 1)
        return out


class Transition(nn.Module):
    def __init__(self, in_planes, out_planes):
        super(Transition, self).__init__()
        self.bn = nn.BatchNorm2d(in_planes)
        self.conv = nn.Conv2d(in_planes, out_planes, kernel_size=1, bias=False)

    def forward(self, x):
        out = self.conv(F.relu(self.bn(x)))
        out = F.avg_pool2d(out, 2)
        return out

class DenseNet(nn.Module):
    def __init__(self, block, nblocks, growth_rate=12, reduction=0.5, num_classes=10):
        super().__init__()
        self.growth_rate = growth_rate

        num_planes = 2*growth_rate
        self.conv1 = nn.Conv2d(3, num_planes, kernel_size=3, padding=1, bias=False)

        self.dense1 = self._make_dense_layers(block, num_planes, nblocks[0])
        num_planes += nblocks[0]*growth_rate
        out_planes = int(math.floor(num_planes*reduction))
        self.trans1 = Transition(num_planes, out_planes)
        num_planes = out_planes

        self.dense2 = self._make_dense_layers(block, num_planes, nblocks[1])
        num_planes += nblocks[1]*growth_rate
        out_planes = int(math.floor(num_planes*reduction))
        self.trans2 = Transition(num_planes, out_planes)
        num_planes = out_planes

        self.dense3 = self._make_dense_layers(block, num_planes, nblocks[2])
        num_planes += nblocks[2]*growth_rate
        out_planes = int(math.floor(num_planes*reduction))
        self.trans3 = Transition(num_planes, out_planes)
        num_planes = out_planes

        self.dense4 = self._make_dense_layers(block, num_planes, nblocks[3])
        num_planes += nblocks[3]*growth_rate

        self.bn = nn.BatchNorm2d(num_planes)
        self.linear = nn.Linear(num_planes, num_classes)

    def _make_dense_layers(self, block, in_planes, nblock):
        layers = []
        for i in range(nblock):
            layers.append(block(in_planes, self.growth_rate))
            in_planes += self.growth_rate
        return nn.Sequential(*layers)

    def forward(self, x):
        out = self.conv1(x)
        out = self.trans1(self.dense1(out))
        out = self.trans2(self.dense2(out))
        out = self.trans3(self.dense3(out))
        out = self.dense4(out)
        out = F.avg_pool2d(F.relu(self.bn(out)), 4)
        out = out.view(out.size(0), -1)
        out = self.linear(out)
        return out

def DenseNet121(num_classes):
    return DenseNet(Bottleneck_DenseNet, [6,12,24,16], growth_rate=32,
        num_classes=num_classes)

def DenseNet169():
    return DenseNet(Bottleneck_DenseNet, [6,12,32,32], growth_rate=32)

def DenseNet201():
    return DenseNet(Bottleneck_DenseNet, [6,12,48,32], growth_rate=32)

def DenseNet161():
    return DenseNet(Bottleneck_DenseNet, [6,12,36,24], growth_rate=48)

def densenet_cifar():
    return DenseNet(Bottleneck_DenseNet, [6,12,24,16], growth_rate=12)

def test():
    net = densenet_cifar()
    x = torch.randn(1,3,32,32)
    y = net(x)
    print(y)


