import transformers

from ..functions import t5_rms_norm_forward, nn_linear_forward, llama_mlp_forward,t5_rms_norm_ow_forward
from ..patch import _checkpoint_module, _patch_module
from ..functions import llama_ckpt_mlp_forward
from ..functions import llama3_selfattn_foward

def apply_patch_to_llama_model(
    model: "transformers.models.llama.modeling_llama.LlamaPreTrainedModel",
    norm: bool = False,
    norm_ow: bool = False,
    attn_in: bool = False,
    attn_out: bool = False,
    mlp_in: bool = False,
    mlp_out: bool = False,
    act_fn: bool = False,
    ckpt_attn: bool = False,
    ckpt_mlp: bool = False,
    ckpt_layer: bool = False,
    soft_max: bool = False,
    ckpt_swiglu:bool = False,
    compress_kwargs: dict | None = None,
) -> None:
    from transformers.models.llama.modeling_llama import LlamaModel, LlamaDecoderLayer
    base_model: LlamaModel = model.base_model

    for layer in base_model.layers:
        layer: LlamaDecoderLayer
        if norm:
            _patch_module(layer.input_layernorm, t5_rms_norm_forward, compress_kwargs=compress_kwargs)
            _patch_module(layer.post_attention_layernorm, t5_rms_norm_forward, compress_kwargs=compress_kwargs)
        if norm_ow:
            _patch_module(layer.input_layernorm, t5_rms_norm_ow_forward, compress_kwargs=compress_kwargs)
            _patch_module(layer.post_attention_layernorm, t5_rms_norm_ow_forward, compress_kwargs=compress_kwargs)       
        if attn_in:
            _patch_module(layer.self_attn.q_proj, nn_linear_forward, compress_kwargs=compress_kwargs)
            _patch_module(layer.self_attn.k_proj, nn_linear_forward, compress_kwargs=compress_kwargs)
            _patch_module(layer.self_attn.v_proj, nn_linear_forward, compress_kwargs=compress_kwargs)
        if attn_out:
            _patch_module(layer.self_attn.o_proj, nn_linear_forward, compress_kwargs=compress_kwargs)
        if mlp_in:
            _patch_module(layer.mlp.gate_proj, nn_linear_forward, compress_kwargs=compress_kwargs)
            _patch_module(layer.mlp.up_proj, nn_linear_forward, compress_kwargs=compress_kwargs)
        if mlp_out:
            _patch_module(layer.mlp.down_proj, nn_linear_forward, compress_kwargs=compress_kwargs)
        if act_fn:
            _patch_module(layer.mlp, llama_mlp_forward, compress_kwargs=compress_kwargs)
        if ckpt_attn:
            _checkpoint_module(layer.self_attn, compress_kwargs=compress_kwargs)
        if ckpt_mlp:
            _checkpoint_module(layer.mlp, compress_kwargs=compress_kwargs)
        if ckpt_layer:
            _checkpoint_module(layer, compress_kwargs=compress_kwargs)
        if ckpt_swiglu:
            _patch_module(layer.mlp, llama_ckpt_mlp_forward, compress_kwargs=compress_kwargs)
        if soft_max:
            _patch_module(layer.self_attn, llama3_selfattn_foward, compress_kwargs=compress_kwargs)
    if norm:
        _patch_module(base_model.norm, t5_rms_norm_forward, compress_kwargs=compress_kwargs)
