radames/Text2Human-API
1
1name: sampler2use_tb_logger: true3set_CUDA_VISIBLE_DEVICES: ~4gpu_ids: [3]5 6# dataset configs7batch_size: 48num_workers: 19train_img_dir: ./datasets/train_images10test_img_dir: ./datasets/test_images11segm_dir: ./datasets/segm12pose_dir: ./datasets/densepose13train_ann_file: ./datasets/texture_ann/train14val_ann_file: ./datasets/texture_ann/val15test_ann_file: ./datasets/texture_ann/test16downsample_factor: 217 18# pretrained models19img_ae_path: ./pretrained_models/vqvae_top.pth20segm_ae_path: ./pretrained_models/parsing_token.pth21 22model_type: TransformerTextureAwareModel23# network configs24 25# image autoencoder26img_embed_dim: 25627img_n_embed: 102428img_double_z: false29img_z_channels: 25630img_resolution: 51231img_in_channels: 332img_out_ch: 333img_ch: 12834img_ch_mult: [1, 1, 2, 2, 4]35img_num_res_blocks: 236img_attn_resolutions: [32]37img_dropout: 0.038 39# segmentation tokenization40segm_double_z: false41segm_z_channels: 3242segm_resolution: 51243segm_in_channels: 2444segm_out_ch: 2445segm_ch: 6446segm_ch_mult: [1, 1, 2, 2, 4]47segm_num_res_blocks: 148segm_attn_resolutions: [16]49segm_dropout: 0.050segm_num_segm_classes: 2451segm_n_embed: 102452segm_embed_dim: 3253 54# sampler configs55codebook_size: 1843256segm_codebook_size: 102457texture_codebook_size: 1858bert_n_emb: 51259bert_n_layers: 2460bert_n_head: 861block_size: 512 # 32 x 1662latent_shape: [32, 16]63embd_pdrop: 0.064resid_pdrop: 0.065attn_pdrop: 0.066num_head: 1867 68# loss configs69loss_type: reweighted_elbo70mask_schedule: random71 72sample_steps: 25673 74# training configs75val_freq: 576print_freq: 10077weight_decay: 078manual_seed: 202179num_epochs: 10080lr: !!float 1e-481lr_decay: step82gamma: 1.083step: 5084 