radames/Text2Human-API
1
1name: sample_from_pose2use_tb_logger: true3set_CUDA_VISIBLE_DEVICES: ~4gpu_ids: [3]5 6# dataset configs7batch_size: 48num_workers: 49pose_dir: ./datasets/densepose10texture_ann_file: ./datasets/texture_ann/test11shape_ann_path: ./datasets/shape_ann/test_ann_file.txt12downsample_factor: 213 14model_type: SampleFromPoseModel15# network configs16embed_dim: 25617n_embed: 102418codebook_spatial_size: 219 20# bottom level vqgan21bot_n_embed: 51222bot_codebook_spatial_size: 223bot_double_z: false24bot_z_channels: 25625bot_resolution: 51226bot_in_channels: 327bot_out_ch: 328bot_ch: 12829bot_ch_mult: [1, 1, 2, 4]30bot_num_res_blocks: 231bot_attn_resolutions: [64]32bot_dropout: 0.033bot_vae_path: ./pretrained_models/vqvae_bottom.pth34 35# top level vqgan36top_double_z: false37top_z_channels: 25638top_resolution: 51239top_in_channels: 340top_out_ch: 341top_ch: 12842top_ch_mult: [1, 1, 2, 2, 4]43top_num_res_blocks: 244top_attn_resolutions: [32]45top_dropout: 0.046top_vae_path: ./pretrained_models/vqvae_top.pth47 48# unet configs49index_pred_encoder_in_channels: 25650index_pred_fc_in_channels: 6451index_pred_fc_in_index: 452index_pred_fc_channels: 6453index_pred_fc_num_convs: 154index_pred_fc_concat_input: False55index_pred_fc_dropout_ratio: 0.156index_pred_fc_num_classes: 51257index_pred_fc_align_corners: False58pretrained_index_network: ./pretrained_models/index_pred_net.pth59 60# segmentation tokenization61segm_double_z: false62segm_z_channels: 3263segm_resolution: 51264segm_in_channels: 2465segm_out_ch: 2466segm_ch: 6467segm_ch_mult: [1, 1, 2, 2, 4]68segm_num_res_blocks: 169segm_attn_resolutions: [16]70segm_dropout: 0.071segm_num_segm_classes: 2472segm_n_embed: 102473segm_embed_dim: 3274segm_token_path: ./pretrained_models/parsing_token.pth75 76# sampler configs77codebook_size: 1843278segm_codebook_size: 102479texture_codebook_size: 1880bert_n_emb: 51281bert_n_layers: 2482bert_n_head: 883block_size: 512 # 32 x 1684latent_shape: [32, 16]85embd_pdrop: 0.086resid_pdrop: 0.087attn_pdrop: 0.088num_head: 1889pretrained_sampler: ./pretrained_models/sampler.pth90 91# shape network configs92shape_embedder_dim: 893shape_embedder_out_dim: 12894shape_attr_class_num: [2, 4, 6, 5, 4, 3, 5, 5, 3, 2, 2, 2, 2, 2, 2]95shape_encoder_in_channels: 196shape_fc_in_channels: 6497shape_fc_in_index: 498shape_fc_channels: 6499shape_fc_num_convs: 1100shape_fc_concat_input: False101shape_fc_dropout_ratio: 0.1102shape_fc_num_classes: 24103shape_fc_align_corners: False104pretrained_parsing_gen: ./pretrained_models/parsing_gen.pth105 106manual_seed: 2021107sample_steps: 256108 