_BASE_: zero_shot_maskformer_R50_bs16_20k.yaml
MODEL:
  BACKBONE:
    NAME: "build_resnet_deeplab_backbone"
  WEIGHTS: "detectron2://DeepLab/R-103.pkl"
  RESNETS:
    DEPTH: 101
    STEM_TYPE: "deeplab"
    STEM_OUT_CHANNELS: 128
    STRIDE_IN_1X1: False
    OUT_FEATURES: ["res2", "res3", "res4", "res5"]
    # NORM: "SyncBN"
    RES5_MULTI_GRID: [1, 2, 4]
  CLIP_ADAPTER:
    PROMPT_LEARNER: "pomp"
    # for learnable prompt
    PROMPT_DIM: 512
    PROMPT_SHAPE: (16, 0)
    CLIP_MODEL_NAME: "ViT-B/16"
    PROMPT_CHECKPOINT: /home/ubuntu/efs/multimodal-prompt-learning/output/imagenet_21k_o/GPT_tutorial_sampled_softmax/vit_b16_ep20_randaug2_unc1000_16shots/nctx16_cscFalse_ctpend/seed42/prompt_learner/model-best.pth.tar
OUTPUT_DIR: output/voc-11k-15/zero_shot_maskformer_R101c_pomp_bs16_20k