hparams: hidden_dim: 512 nhead: 8 low_rank_dim: 64 query_residual_weight: 0.5 anchor_scale: 0.8 num_steps: 4 num_layers: 6 embedding_dim: 768 use_fro_norm: true feature_extractor: openai/clip-vit-large-patch14