_BASE_: zero_shot_maskformer_R50_single_prompt_bs32_60k.yaml
MODEL:
  META_ARCHITECTURE: "ZeroShotPerPixelModel"
  BACKBONE:
    NAME: "build_resnet_deeplab_backbone"
  WEIGHTS: "detectron2://DeepLab/R-103.pkl"
  RESNETS:
    DEPTH: 101
    STEM_TYPE: "deeplab"
    STEM_OUT_CHANNELS: 128
    STRIDE_IN_1X1: False
    OUT_FEATURES: ["res2", "res3", "res4", "res5"]
    # NORM: "SyncBN"
    RES5_MULTI_GRID: [1, 2, 4]
  SEM_SEG_HEAD:
    NAME: "ZeroPerPixelBaselineHead"
    IN_FEATURES: ["res2", "res3", "res4", "res5"]
    IGNORE_VALUE: 255
    NUM_CLASSES: 512
    COMMON_STRIDE: 4  # not used, hard-coded
    LOSS_WEIGHT: 1.0
    CONVS_DIM: 256
    MASK_DIM: 256
    NORM: "GN"
  CLIP_ADAPTER:
    CLIP_ENSEMBLE_WEIGHT: 0.7