{ "n_obs_steps": 1, "chunk_size": 100, "n_action_steps": 100, "input_shapes": { "observation.images.top": [ 3, 480, 640 ], "observation.state": [ 14 ] }, "output_shapes": { "action": [ 14 ] }, "input_normalization_modes": { "observation.images.top": "mean_std", "observation.state": "mean_std" }, "output_normalization_modes": { "action": "mean_std" }, "vision_backbone": "resnet18", "pretrained_backbone_weights": "ResNet18_Weights.IMAGENET1K_V1", "replace_final_stride_with_dilation": false, "pre_norm": false, "dim_model": 512, "n_heads": 8, "dim_feedforward": 3200, "feedforward_activation": "relu", "n_encoder_layers": 4, "n_decoder_layers": 1, "use_vae": true, "latent_dim": 32, "n_vae_encoder_layers": 4, "use_temporal_aggregation": false, "dropout": 0.1, "kl_weight": 10.0 }