{
    "embed_dim": 512,
    "vision_cfg": {
        "layers": 14,
        "width": 768,
        "patch_size": 16,
        "image_size": 256,
        "num_heads": 12,
        "heavy_conv": true
    },
    "text_cfg": {
        "context_length": 77,
        "vocab_size": 49408,
        "width": 512,
        "heads": 8,
        "layers": 12
    },
    "latent_token_num": 32,
    "custom_vision": true
}