"""
Asymmetric Loss for Multi-label Classification

Implementation of Asymmetric Loss from:
"Asymmetric Loss For Multi-Label Classification"
https://arxiv.org/abs/2009.14119

This loss is specifically designed for multi-label classification where
positive and negative samples should be treated differently.
"""

import torch
import torch.nn as nn
import torch.nn.functional as F


class AsymmetricLoss(nn.Module):
    """
    Asymmetric Loss for multi-label classification
    
    This loss applies different focusing parameters to positive and negative samples:
    - Positive samples: gamma_pos (usually 0, no focusing)
    - Negative samples: gamma_neg (usually 4, focus on hard negatives)
    """
    
    def __init__(self, 
                 gamma_neg: float = 4.0,
                 gamma_pos: float = 0.0, 
                 clip: float = 0.05,
                 eps: float = 1e-8):
        """
        Initialize Asymmetric Loss
        
        Args:
            gamma_neg: Focusing parameter for negative samples (higher = focus more on hard negatives)
            gamma_pos: Focusing parameter for positive samples (usually 0)
            clip: Probability clipping for numerical stability
            eps: Small epsilon to prevent log(0)
        """
        super().__init__()
        self.gamma_neg = gamma_neg
        self.gamma_pos = gamma_pos
        self.clip = clip
        self.eps = eps
        
    def forward(self, x: torch.Tensor, y: torch.Tensor) -> torch.Tensor:
        """
        Compute asymmetric loss
        
        Args:
            x: Logits [batch_size, num_classes]
            y: Binary targets [batch_size, num_classes]
            
        Returns:
            loss: Asymmetric loss value
        """
        # Convert logits to probabilities
        x_sigmoid = torch.sigmoid(x)
        
        # Positive and negative probabilities
        x_pos = x_sigmoid
        x_neg = 1 - x_sigmoid
        
        # Probability clipping for numerical stability
        if self.clip is not None and self.clip > 0:
            x_neg = (x_neg + self.clip).clamp(max=1)
            
        # Asymmetric focusing weights
        pos_weight = (1 - x_pos) ** self.gamma_pos
        neg_weight = x_pos ** self.gamma_neg
        
        # Loss calculation
        loss_pos = y * torch.log(x_pos.clamp(min=self.eps)) * pos_weight
        loss_neg = (1 - y) * torch.log(x_neg.clamp(min=self.eps)) * neg_weight
        
        loss = -(loss_pos + loss_neg)
        
        # Return mean loss
        return loss.mean()
    
    def __repr__(self):
        return f"AsymmetricLoss(gamma_neg={self.gamma_neg}, gamma_pos={self.gamma_pos}, clip={self.clip})"


class AsymmetricLossOptimized(nn.Module):
    """
    Optimized version of Asymmetric Loss with better numerical stability
    """
    
    def __init__(self, 
                 gamma_neg: float = 4.0,
                 gamma_pos: float = 0.0,
                 clip: float = 0.05,
                 disable_torch_grad_focal_loss: bool = False):
        super().__init__()
        self.gamma_neg = gamma_neg
        self.gamma_pos = gamma_pos
        self.clip = clip
        self.disable_torch_grad_focal_loss = disable_torch_grad_focal_loss
        
    def forward(self, x, y):
        """"
        Parameters
        ----------
        x: input logits
        y: targets (multi-label binarized vector)
        """
        # Calculating Probabilities
        x_sigmoid = torch.sigmoid(x)
        xs_pos = x_sigmoid
        xs_neg = 1 - x_sigmoid

        # Asymmetric Clipping
        if self.clip is not None and self.clip > 0:
            xs_neg = (xs_neg + self.clip).clamp(max=1)

        # Basic CE calculation
        los_pos = y * torch.log(xs_pos.clamp(min=1e-8))
        los_neg = (1 - y) * torch.log(xs_neg.clamp(min=1e-8))
        loss = los_pos + los_neg

        # Asymmetric Focusing
        if self.gamma_neg > 0 or self.gamma_pos > 0:
            if self.disable_torch_grad_focal_loss:
                torch.set_grad_enabled(False)
            pt0 = xs_pos * y
            pt1 = xs_neg * (1 - y)  # pt = p if t > 0 else 1-p
            pt = pt0 + pt1
            one_sided_gamma = self.gamma_pos * y + self.gamma_neg * (1 - y)
            one_sided_w = torch.pow(1 - pt, one_sided_gamma)
            if self.disable_torch_grad_focal_loss:
                torch.set_grad_enabled(True)
            loss *= one_sided_w

        return -loss.mean()


# For convenience, use the optimized version by default
AsymmetricLoss = AsymmetricLossOptimized


class CombinedLoss(nn.Module):
    """
    Combined Loss: Asymmetric Loss + BCE Loss
    
    This combines the ranking benefits of Asymmetric Loss with the 
    probability calibration benefits of BCE Loss.
    """
    
    def __init__(self, 
                 alpha: float = 0.7,
                 beta: float = 0.3,
                 gamma_neg: float = 4.0,
                 gamma_pos: float = 0.0,
                 clip: float = 0.05):
        """
        Initialize Combined Loss
        
        Args:
            alpha: Weight for Asymmetric Loss (ranking quality)
            beta: Weight for BCE Loss (probability calibration)
            gamma_neg: Asymmetric loss gamma for negative samples
            gamma_pos: Asymmetric loss gamma for positive samples  
            clip: Asymmetric loss probability clipping
        """
        super().__init__()
        self.alpha = alpha
        self.beta = beta
        
        # Ensure weights sum to 1
        total_weight = alpha + beta
        self.alpha = alpha / total_weight
        self.beta = beta / total_weight
        
        self.asymmetric_loss = AsymmetricLossOptimized(
            gamma_neg=gamma_neg,
            gamma_pos=gamma_pos,
            clip=clip
        )
        self.bce_loss = nn.BCEWithLogitsLoss()
        
        print(f"CombinedLoss initialized: α={self.alpha:.3f} (Asymmetric) + β={self.beta:.3f} (BCE)")
    
    def forward(self, logits: torch.Tensor, targets: torch.Tensor) -> torch.Tensor:
        """
        Compute combined loss
        
        Args:
            logits: Model predictions [batch_size, num_classes]
            targets: Ground truth labels [batch_size, num_classes]
            
        Returns:
            Combined loss value
        """
        asym_loss = self.asymmetric_loss(logits, targets)
        bce_loss = self.bce_loss(logits, targets)
        
        combined_loss = self.alpha * asym_loss + self.beta * bce_loss
        
        return combined_loss
    
    def get_individual_losses(self, logits: torch.Tensor, targets: torch.Tensor):
        """
        Get individual loss components for logging
        
        Returns:
            tuple: (asymmetric_loss, bce_loss, combined_loss)
        """
        asym_loss = self.asymmetric_loss(logits, targets)
        bce_loss = self.bce_loss(logits, targets)
        combined_loss = self.alpha * asym_loss + self.beta * bce_loss
        
        return asym_loss.item(), bce_loss.item(), combined_loss
    
    def __repr__(self):
        return f"CombinedLoss(α={self.alpha:.3f}, β={self.beta:.3f}, γ_neg={self.asymmetric_loss.gamma_neg})"