Arulkumar03/Wheat_HEAD_Detection_Counting_ComputerVision_Model
0
1# Copyright (c) Facebook, Inc. and its affiliates.2from .config import CfgNode as CN3 4# NOTE: given the new config system5# (https://detectron2.readthedocs.io/en/latest/tutorials/lazyconfigs.html),6# we will stop adding new functionalities to default CfgNode.7 8# -----------------------------------------------------------------------------9# Convention about Training / Test specific parameters10# -----------------------------------------------------------------------------11# Whenever an argument can be either used for training or for testing, the12# corresponding name will be post-fixed by a _TRAIN for a training parameter,13# or _TEST for a test-specific parameter.14# For example, the number of images during training will be15# IMAGES_PER_BATCH_TRAIN, while the number of images for testing will be16# IMAGES_PER_BATCH_TEST17 18# -----------------------------------------------------------------------------19# Config definition20# -----------------------------------------------------------------------------21 22_C = CN()23 24# The version number, to upgrade from old configs to new ones if any25# changes happen. It's recommended to keep a VERSION in your config file.26_C.VERSION = 227 28_C.MODEL = CN()29_C.MODEL.LOAD_PROPOSALS = False30_C.MODEL.MASK_ON = False31_C.MODEL.KEYPOINT_ON = False32_C.MODEL.DEVICE = "cuda"33_C.MODEL.META_ARCHITECTURE = "GeneralizedRCNN"34 35# Path (a file path, or URL like detectron2://.., https://..) to a checkpoint file36# to be loaded to the model. You can find available models in the model zoo.37_C.MODEL.WEIGHTS = ""38 39# Values to be used for image normalization (BGR order, since INPUT.FORMAT defaults to BGR).40# To train on images of different number of channels, just set different mean & std.41# Default values are the mean pixel value from ImageNet: [103.53, 116.28, 123.675]42_C.MODEL.PIXEL_MEAN = [103.530, 116.280, 123.675]43# When using pre-trained models in Detectron1 or any MSRA models,44# std has been absorbed into its conv1 weights, so the std needs to be set 1.45# Otherwise, you can use [57.375, 57.120, 58.395] (ImageNet std)46_C.MODEL.PIXEL_STD = [1.0, 1.0, 1.0]47 48 49# -----------------------------------------------------------------------------50# INPUT51# -----------------------------------------------------------------------------52_C.INPUT = CN()53# By default, {MIN,MAX}_SIZE options are used in transforms.ResizeShortestEdge.54# Please refer to ResizeShortestEdge for detailed definition.55# Size of the smallest side of the image during training56_C.INPUT.MIN_SIZE_TRAIN = (800,)57# Sample size of smallest side by choice or random selection from range give by58# INPUT.MIN_SIZE_TRAIN59_C.INPUT.MIN_SIZE_TRAIN_SAMPLING = "choice"60# Maximum size of the side of the image during training61_C.INPUT.MAX_SIZE_TRAIN = 133362# Size of the smallest side of the image during testing. Set to zero to disable resize in testing.63_C.INPUT.MIN_SIZE_TEST = 80064# Maximum size of the side of the image during testing65_C.INPUT.MAX_SIZE_TEST = 133366# Mode for flipping images used in data augmentation during training67# choose one of ["horizontal, "vertical", "none"]68_C.INPUT.RANDOM_FLIP = "horizontal"69 70# `True` if cropping is used for data augmentation during training71_C.INPUT.CROP = CN({"ENABLED": False})72# Cropping type. See documentation of `detectron2.data.transforms.RandomCrop` for explanation.73_C.INPUT.CROP.TYPE = "relative_range"74# Size of crop in range (0, 1] if CROP.TYPE is "relative" or "relative_range" and in number of75# pixels if CROP.TYPE is "absolute"76_C.INPUT.CROP.SIZE = [0.9, 0.9]77 78 79# Whether the model needs RGB, YUV, HSV etc.80# Should be one of the modes defined here, as we use PIL to read the image:81# https://pillow.readthedocs.io/en/stable/handbook/concepts.html#concept-modes82# with BGR being the one exception. One can set image format to BGR, we will83# internally use RGB for conversion and flip the channels over84_C.INPUT.FORMAT = "BGR"85# The ground truth mask format that the model will use.86# Mask R-CNN supports either "polygon" or "bitmask" as ground truth.87_C.INPUT.MASK_FORMAT = "polygon" # alternative: "bitmask"88 89 90# -----------------------------------------------------------------------------91# Dataset92# -----------------------------------------------------------------------------93_C.DATASETS = CN()94# List of the dataset names for training. Must be registered in DatasetCatalog95# Samples from these datasets will be merged and used as one dataset.96_C.DATASETS.TRAIN = ()97# List of the pre-computed proposal files for training, which must be consistent98# with datasets listed in DATASETS.TRAIN.99_C.DATASETS.PROPOSAL_FILES_TRAIN = ()100# Number of top scoring precomputed proposals to keep for training101_C.DATASETS.PRECOMPUTED_PROPOSAL_TOPK_TRAIN = 2000102# List of the dataset names for testing. Must be registered in DatasetCatalog103_C.DATASETS.TEST = ()104# List of the pre-computed proposal files for test, which must be consistent105# with datasets listed in DATASETS.TEST.106_C.DATASETS.PROPOSAL_FILES_TEST = ()107# Number of top scoring precomputed proposals to keep for test108_C.DATASETS.PRECOMPUTED_PROPOSAL_TOPK_TEST = 1000109 110# -----------------------------------------------------------------------------111# DataLoader112# -----------------------------------------------------------------------------113_C.DATALOADER = CN()114# Number of data loading threads115_C.DATALOADER.NUM_WORKERS = 4116# If True, each batch should contain only images for which the aspect ratio117# is compatible. This groups portrait images together, and landscape images118# are not batched with portrait images.119_C.DATALOADER.ASPECT_RATIO_GROUPING = True120# Options: TrainingSampler, RepeatFactorTrainingSampler121_C.DATALOADER.SAMPLER_TRAIN = "TrainingSampler"122# Repeat threshold for RepeatFactorTrainingSampler123_C.DATALOADER.REPEAT_THRESHOLD = 0.0124# Tf True, when working on datasets that have instance annotations, the125# training dataloader will filter out images without associated annotations126_C.DATALOADER.FILTER_EMPTY_ANNOTATIONS = True127 128# ---------------------------------------------------------------------------- #129# Backbone options130# ---------------------------------------------------------------------------- #131_C.MODEL.BACKBONE = CN()132 133_C.MODEL.BACKBONE.NAME = "build_resnet_backbone"134# Freeze the first several stages so they are not trained.135# There are 5 stages in ResNet. The first is a convolution, and the following136# stages are each group of residual blocks.137_C.MODEL.BACKBONE.FREEZE_AT = 2138 139 140# ---------------------------------------------------------------------------- #141# FPN options142# ---------------------------------------------------------------------------- #143_C.MODEL.FPN = CN()144# Names of the input feature maps to be used by FPN145# They must have contiguous power of 2 strides146# e.g., ["res2", "res3", "res4", "res5"]147_C.MODEL.FPN.IN_FEATURES = []148_C.MODEL.FPN.OUT_CHANNELS = 256149 150# Options: "" (no norm), "GN"151_C.MODEL.FPN.NORM = ""152 153# Types for fusing the FPN top-down and lateral features. Can be either "sum" or "avg"154_C.MODEL.FPN.FUSE_TYPE = "sum"155 156 157# ---------------------------------------------------------------------------- #158# Proposal generator options159# ---------------------------------------------------------------------------- #160_C.MODEL.PROPOSAL_GENERATOR = CN()161# Current proposal generators include "RPN", "RRPN" and "PrecomputedProposals"162_C.MODEL.PROPOSAL_GENERATOR.NAME = "RPN"163# Proposal height and width both need to be greater than MIN_SIZE164# (a the scale used during training or inference)165_C.MODEL.PROPOSAL_GENERATOR.MIN_SIZE = 0166 167 168# ---------------------------------------------------------------------------- #169# Anchor generator options170# ---------------------------------------------------------------------------- #171_C.MODEL.ANCHOR_GENERATOR = CN()172# The generator can be any name in the ANCHOR_GENERATOR registry173_C.MODEL.ANCHOR_GENERATOR.NAME = "DefaultAnchorGenerator"174# Anchor sizes (i.e. sqrt of area) in absolute pixels w.r.t. the network input.175# Format: list[list[float]]. SIZES[i] specifies the list of sizes to use for176# IN_FEATURES[i]; len(SIZES) must be equal to len(IN_FEATURES) or 1.177# When len(SIZES) == 1, SIZES[0] is used for all IN_FEATURES.178_C.MODEL.ANCHOR_GENERATOR.SIZES = [[32, 64, 128, 256, 512]]179# Anchor aspect ratios. For each area given in `SIZES`, anchors with different aspect180# ratios are generated by an anchor generator.181# Format: list[list[float]]. ASPECT_RATIOS[i] specifies the list of aspect ratios (H/W)182# to use for IN_FEATURES[i]; len(ASPECT_RATIOS) == len(IN_FEATURES) must be true,183# or len(ASPECT_RATIOS) == 1 is true and aspect ratio list ASPECT_RATIOS[0] is used184# for all IN_FEATURES.185_C.MODEL.ANCHOR_GENERATOR.ASPECT_RATIOS = [[0.5, 1.0, 2.0]]186# Anchor angles.187# list[list[float]], the angle in degrees, for each input feature map.188# ANGLES[i] specifies the list of angles for IN_FEATURES[i].189_C.MODEL.ANCHOR_GENERATOR.ANGLES = [[-90, 0, 90]]190# Relative offset between the center of the first anchor and the top-left corner of the image191# Value has to be in [0, 1). Recommend to use 0.5, which means half stride.192# The value is not expected to affect model accuracy.193_C.MODEL.ANCHOR_GENERATOR.OFFSET = 0.0194 195# ---------------------------------------------------------------------------- #196# RPN options197# ---------------------------------------------------------------------------- #198_C.MODEL.RPN = CN()199_C.MODEL.RPN.HEAD_NAME = "StandardRPNHead" # used by RPN_HEAD_REGISTRY200 201# Names of the input feature maps to be used by RPN202# e.g., ["p2", "p3", "p4", "p5", "p6"] for FPN203_C.MODEL.RPN.IN_FEATURES = ["res4"]204# Remove RPN anchors that go outside the image by BOUNDARY_THRESH pixels205# Set to -1 or a large value, e.g. 100000, to disable pruning anchors206_C.MODEL.RPN.BOUNDARY_THRESH = -1207# IOU overlap ratios [BG_IOU_THRESHOLD, FG_IOU_THRESHOLD]208# Minimum overlap required between an anchor and ground-truth box for the209# (anchor, gt box) pair to be a positive example (IoU >= FG_IOU_THRESHOLD210# ==> positive RPN example: 1)211# Maximum overlap allowed between an anchor and ground-truth box for the212# (anchor, gt box) pair to be a negative examples (IoU < BG_IOU_THRESHOLD213# ==> negative RPN example: 0)214# Anchors with overlap in between (BG_IOU_THRESHOLD <= IoU < FG_IOU_THRESHOLD)215# are ignored (-1)216_C.MODEL.RPN.IOU_THRESHOLDS = [0.3, 0.7]217_C.MODEL.RPN.IOU_LABELS = [0, -1, 1]218# Number of regions per image used to train RPN219_C.MODEL.RPN.BATCH_SIZE_PER_IMAGE = 256220# Target fraction of foreground (positive) examples per RPN minibatch221_C.MODEL.RPN.POSITIVE_FRACTION = 0.5222# Options are: "smooth_l1", "giou", "diou", "ciou"223_C.MODEL.RPN.BBOX_REG_LOSS_TYPE = "smooth_l1"224_C.MODEL.RPN.BBOX_REG_LOSS_WEIGHT = 1.0225# Weights on (dx, dy, dw, dh) for normalizing RPN anchor regression targets226_C.MODEL.RPN.BBOX_REG_WEIGHTS = (1.0, 1.0, 1.0, 1.0)227# The transition point from L1 to L2 loss. Set to 0.0 to make the loss simply L1.228_C.MODEL.RPN.SMOOTH_L1_BETA = 0.0229_C.MODEL.RPN.LOSS_WEIGHT = 1.0230# Number of top scoring RPN proposals to keep before applying NMS231# When FPN is used, this is *per FPN level* (not total)232_C.MODEL.RPN.PRE_NMS_TOPK_TRAIN = 12000233_C.MODEL.RPN.PRE_NMS_TOPK_TEST = 6000234# Number of top scoring RPN proposals to keep after applying NMS235# When FPN is used, this limit is applied per level and then again to the union236# of proposals from all levels237# NOTE: When FPN is used, the meaning of this config is different from Detectron1.238# It means per-batch topk in Detectron1, but per-image topk here.239# See the "find_top_rpn_proposals" function for details.240_C.MODEL.RPN.POST_NMS_TOPK_TRAIN = 2000241_C.MODEL.RPN.POST_NMS_TOPK_TEST = 1000242# NMS threshold used on RPN proposals243_C.MODEL.RPN.NMS_THRESH = 0.7244# Set this to -1 to use the same number of output channels as input channels.245_C.MODEL.RPN.CONV_DIMS = [-1]246 247# ---------------------------------------------------------------------------- #248# ROI HEADS options249# ---------------------------------------------------------------------------- #250_C.MODEL.ROI_HEADS = CN()251_C.MODEL.ROI_HEADS.NAME = "Res5ROIHeads"252# Number of foreground classes253_C.MODEL.ROI_HEADS.NUM_CLASSES = 80254# Names of the input feature maps to be used by ROI heads255# Currently all heads (box, mask, ...) use the same input feature map list256# e.g., ["p2", "p3", "p4", "p5"] is commonly used for FPN257_C.MODEL.ROI_HEADS.IN_FEATURES = ["res4"]258# IOU overlap ratios [IOU_THRESHOLD]259# Overlap threshold for an RoI to be considered background (if < IOU_THRESHOLD)260# Overlap threshold for an RoI to be considered foreground (if >= IOU_THRESHOLD)261_C.MODEL.ROI_HEADS.IOU_THRESHOLDS = [0.5]262_C.MODEL.ROI_HEADS.IOU_LABELS = [0, 1]263# RoI minibatch size *per image* (number of regions of interest [ROIs]) during training264# Total number of RoIs per training minibatch =265# ROI_HEADS.BATCH_SIZE_PER_IMAGE * SOLVER.IMS_PER_BATCH266# E.g., a common configuration is: 512 * 16 = 8192267_C.MODEL.ROI_HEADS.BATCH_SIZE_PER_IMAGE = 512268# Target fraction of RoI minibatch that is labeled foreground (i.e. class > 0)269_C.MODEL.ROI_HEADS.POSITIVE_FRACTION = 0.25270 271# Only used on test mode272 273# Minimum score threshold (assuming scores in a [0, 1] range); a value chosen to274# balance obtaining high recall with not having too many low precision275# detections that will slow down inference post processing steps (like NMS)276# A default threshold of 0.0 increases AP by ~0.2-0.3 but significantly slows down277# inference.278_C.MODEL.ROI_HEADS.SCORE_THRESH_TEST = 0.05279# Overlap threshold used for non-maximum suppression (suppress boxes with280# IoU >= this threshold)281_C.MODEL.ROI_HEADS.NMS_THRESH_TEST = 0.5282# If True, augment proposals with ground-truth boxes before sampling proposals to283# train ROI heads.284_C.MODEL.ROI_HEADS.PROPOSAL_APPEND_GT = True285 286# ---------------------------------------------------------------------------- #287# Box Head288# ---------------------------------------------------------------------------- #289_C.MODEL.ROI_BOX_HEAD = CN()290# C4 don't use head name option291# Options for non-C4 models: FastRCNNConvFCHead,292_C.MODEL.ROI_BOX_HEAD.NAME = ""293# Options are: "smooth_l1", "giou", "diou", "ciou"294_C.MODEL.ROI_BOX_HEAD.BBOX_REG_LOSS_TYPE = "smooth_l1"295# The final scaling coefficient on the box regression loss, used to balance the magnitude of its296# gradients with other losses in the model. See also `MODEL.ROI_KEYPOINT_HEAD.LOSS_WEIGHT`.297_C.MODEL.ROI_BOX_HEAD.BBOX_REG_LOSS_WEIGHT = 1.0298# Default weights on (dx, dy, dw, dh) for normalizing bbox regression targets299# These are empirically chosen to approximately lead to unit variance targets300_C.MODEL.ROI_BOX_HEAD.BBOX_REG_WEIGHTS = (10.0, 10.0, 5.0, 5.0)301# The transition point from L1 to L2 loss. Set to 0.0 to make the loss simply L1.302_C.MODEL.ROI_BOX_HEAD.SMOOTH_L1_BETA = 0.0303_C.MODEL.ROI_BOX_HEAD.POOLER_RESOLUTION = 14304_C.MODEL.ROI_BOX_HEAD.POOLER_SAMPLING_RATIO = 0305# Type of pooling operation applied to the incoming feature map for each RoI306_C.MODEL.ROI_BOX_HEAD.POOLER_TYPE = "ROIAlignV2"307 308_C.MODEL.ROI_BOX_HEAD.NUM_FC = 0309# Hidden layer dimension for FC layers in the RoI box head310_C.MODEL.ROI_BOX_HEAD.FC_DIM = 1024311_C.MODEL.ROI_BOX_HEAD.NUM_CONV = 0312# Channel dimension for Conv layers in the RoI box head313_C.MODEL.ROI_BOX_HEAD.CONV_DIM = 256314# Normalization method for the convolution layers.315# Options: "" (no norm), "GN", "SyncBN".316_C.MODEL.ROI_BOX_HEAD.NORM = ""317# Whether to use class agnostic for bbox regression318_C.MODEL.ROI_BOX_HEAD.CLS_AGNOSTIC_BBOX_REG = False319# If true, RoI heads use bounding boxes predicted by the box head rather than proposal boxes.320_C.MODEL.ROI_BOX_HEAD.TRAIN_ON_PRED_BOXES = False321 322# Federated loss can be used to improve the training of LVIS323_C.MODEL.ROI_BOX_HEAD.USE_FED_LOSS = False324# Sigmoid cross entrophy is used with federated loss325_C.MODEL.ROI_BOX_HEAD.USE_SIGMOID_CE = False326# The power value applied to image_count when calcualting frequency weight327_C.MODEL.ROI_BOX_HEAD.FED_LOSS_FREQ_WEIGHT_POWER = 0.5328# Number of classes to keep in total329_C.MODEL.ROI_BOX_HEAD.FED_LOSS_NUM_CLASSES = 50330 331# ---------------------------------------------------------------------------- #332# Cascaded Box Head333# ---------------------------------------------------------------------------- #334_C.MODEL.ROI_BOX_CASCADE_HEAD = CN()335# The number of cascade stages is implicitly defined by the length of the following two configs.336_C.MODEL.ROI_BOX_CASCADE_HEAD.BBOX_REG_WEIGHTS = (337 (10.0, 10.0, 5.0, 5.0),338 (20.0, 20.0, 10.0, 10.0),339 (30.0, 30.0, 15.0, 15.0),340)341_C.MODEL.ROI_BOX_CASCADE_HEAD.IOUS = (0.5, 0.6, 0.7)342 343 344# ---------------------------------------------------------------------------- #345# Mask Head346# ---------------------------------------------------------------------------- #347_C.MODEL.ROI_MASK_HEAD = CN()348_C.MODEL.ROI_MASK_HEAD.NAME = "MaskRCNNConvUpsampleHead"349_C.MODEL.ROI_MASK_HEAD.POOLER_RESOLUTION = 14350_C.MODEL.ROI_MASK_HEAD.POOLER_SAMPLING_RATIO = 0351_C.MODEL.ROI_MASK_HEAD.NUM_CONV = 0 # The number of convs in the mask head352_C.MODEL.ROI_MASK_HEAD.CONV_DIM = 256353# Normalization method for the convolution layers.354# Options: "" (no norm), "GN", "SyncBN".355_C.MODEL.ROI_MASK_HEAD.NORM = ""356# Whether to use class agnostic for mask prediction357_C.MODEL.ROI_MASK_HEAD.CLS_AGNOSTIC_MASK = False358# Type of pooling operation applied to the incoming feature map for each RoI359_C.MODEL.ROI_MASK_HEAD.POOLER_TYPE = "ROIAlignV2"360 361 362# ---------------------------------------------------------------------------- #363# Keypoint Head364# ---------------------------------------------------------------------------- #365_C.MODEL.ROI_KEYPOINT_HEAD = CN()366_C.MODEL.ROI_KEYPOINT_HEAD.NAME = "KRCNNConvDeconvUpsampleHead"367_C.MODEL.ROI_KEYPOINT_HEAD.POOLER_RESOLUTION = 14368_C.MODEL.ROI_KEYPOINT_HEAD.POOLER_SAMPLING_RATIO = 0369_C.MODEL.ROI_KEYPOINT_HEAD.CONV_DIMS = tuple(512 for _ in range(8))370_C.MODEL.ROI_KEYPOINT_HEAD.NUM_KEYPOINTS = 17 # 17 is the number of keypoints in COCO.371 372# Images with too few (or no) keypoints are excluded from training.373_C.MODEL.ROI_KEYPOINT_HEAD.MIN_KEYPOINTS_PER_IMAGE = 1374# Normalize by the total number of visible keypoints in the minibatch if True.375# Otherwise, normalize by the total number of keypoints that could ever exist376# in the minibatch.377# The keypoint softmax loss is only calculated on visible keypoints.378# Since the number of visible keypoints can vary significantly between379# minibatches, this has the effect of up-weighting the importance of380# minibatches with few visible keypoints. (Imagine the extreme case of381# only one visible keypoint versus N: in the case of N, each one382# contributes 1/N to the gradient compared to the single keypoint383# determining the gradient direction). Instead, we can normalize the384# loss by the total number of keypoints, if it were the case that all385# keypoints were visible in a full minibatch. (Returning to the example,386# this means that the one visible keypoint contributes as much as each387# of the N keypoints.)388_C.MODEL.ROI_KEYPOINT_HEAD.NORMALIZE_LOSS_BY_VISIBLE_KEYPOINTS = True389# Multi-task loss weight to use for keypoints390# Recommended values:391# - use 1.0 if NORMALIZE_LOSS_BY_VISIBLE_KEYPOINTS is True392# - use 4.0 if NORMALIZE_LOSS_BY_VISIBLE_KEYPOINTS is False393_C.MODEL.ROI_KEYPOINT_HEAD.LOSS_WEIGHT = 1.0394# Type of pooling operation applied to the incoming feature map for each RoI395_C.MODEL.ROI_KEYPOINT_HEAD.POOLER_TYPE = "ROIAlignV2"396 397# ---------------------------------------------------------------------------- #398# Semantic Segmentation Head399# ---------------------------------------------------------------------------- #400_C.MODEL.SEM_SEG_HEAD = CN()401_C.MODEL.SEM_SEG_HEAD.NAME = "SemSegFPNHead"402_C.MODEL.SEM_SEG_HEAD.IN_FEATURES = ["p2", "p3", "p4", "p5"]403# Label in the semantic segmentation ground truth that is ignored, i.e., no loss is calculated for404# the correposnding pixel.405_C.MODEL.SEM_SEG_HEAD.IGNORE_VALUE = 255406# Number of classes in the semantic segmentation head407_C.MODEL.SEM_SEG_HEAD.NUM_CLASSES = 54408# Number of channels in the 3x3 convs inside semantic-FPN heads.409_C.MODEL.SEM_SEG_HEAD.CONVS_DIM = 128410# Outputs from semantic-FPN heads are up-scaled to the COMMON_STRIDE stride.411_C.MODEL.SEM_SEG_HEAD.COMMON_STRIDE = 4412# Normalization method for the convolution layers. Options: "" (no norm), "GN".413_C.MODEL.SEM_SEG_HEAD.NORM = "GN"414_C.MODEL.SEM_SEG_HEAD.LOSS_WEIGHT = 1.0415 416_C.MODEL.PANOPTIC_FPN = CN()417# Scaling of all losses from instance detection / segmentation head.418_C.MODEL.PANOPTIC_FPN.INSTANCE_LOSS_WEIGHT = 1.0419 420# options when combining instance & semantic segmentation outputs421_C.MODEL.PANOPTIC_FPN.COMBINE = CN({"ENABLED": True}) # "COMBINE.ENABLED" is deprecated & not used422_C.MODEL.PANOPTIC_FPN.COMBINE.OVERLAP_THRESH = 0.5423_C.MODEL.PANOPTIC_FPN.COMBINE.STUFF_AREA_LIMIT = 4096424_C.MODEL.PANOPTIC_FPN.COMBINE.INSTANCES_CONFIDENCE_THRESH = 0.5425 426 427# ---------------------------------------------------------------------------- #428# RetinaNet Head429# ---------------------------------------------------------------------------- #430_C.MODEL.RETINANET = CN()431 432# This is the number of foreground classes.433_C.MODEL.RETINANET.NUM_CLASSES = 80434 435_C.MODEL.RETINANET.IN_FEATURES = ["p3", "p4", "p5", "p6", "p7"]436 437# Convolutions to use in the cls and bbox tower438# NOTE: this doesn't include the last conv for logits439_C.MODEL.RETINANET.NUM_CONVS = 4440 441# IoU overlap ratio [bg, fg] for labeling anchors.442# Anchors with < bg are labeled negative (0)443# Anchors with >= bg and < fg are ignored (-1)444# Anchors with >= fg are labeled positive (1)445_C.MODEL.RETINANET.IOU_THRESHOLDS = [0.4, 0.5]446_C.MODEL.RETINANET.IOU_LABELS = [0, -1, 1]447 448# Prior prob for rare case (i.e. foreground) at the beginning of training.449# This is used to set the bias for the logits layer of the classifier subnet.450# This improves training stability in the case of heavy class imbalance.451_C.MODEL.RETINANET.PRIOR_PROB = 0.01452 453# Inference cls score threshold, only anchors with score > INFERENCE_TH are454# considered for inference (to improve speed)455_C.MODEL.RETINANET.SCORE_THRESH_TEST = 0.05456# Select topk candidates before NMS457_C.MODEL.RETINANET.TOPK_CANDIDATES_TEST = 1000458_C.MODEL.RETINANET.NMS_THRESH_TEST = 0.5459 460# Weights on (dx, dy, dw, dh) for normalizing Retinanet anchor regression targets461_C.MODEL.RETINANET.BBOX_REG_WEIGHTS = (1.0, 1.0, 1.0, 1.0)462 463# Loss parameters464_C.MODEL.RETINANET.FOCAL_LOSS_GAMMA = 2.0465_C.MODEL.RETINANET.FOCAL_LOSS_ALPHA = 0.25466_C.MODEL.RETINANET.SMOOTH_L1_LOSS_BETA = 0.1467# Options are: "smooth_l1", "giou", "diou", "ciou"468_C.MODEL.RETINANET.BBOX_REG_LOSS_TYPE = "smooth_l1"469 470# One of BN, SyncBN, FrozenBN, GN471# Only supports GN until unshared norm is implemented472_C.MODEL.RETINANET.NORM = ""473 474 475# ---------------------------------------------------------------------------- #476# ResNe[X]t options (ResNets = {ResNet, ResNeXt}477# Note that parts of a resnet may be used for both the backbone and the head478# These options apply to both479# ---------------------------------------------------------------------------- #480_C.MODEL.RESNETS = CN()481 482_C.MODEL.RESNETS.DEPTH = 50483_C.MODEL.RESNETS.OUT_FEATURES = ["res4"] # res4 for C4 backbone, res2..5 for FPN backbone484 485# Number of groups to use; 1 ==> ResNet; > 1 ==> ResNeXt486_C.MODEL.RESNETS.NUM_GROUPS = 1487 488# Options: FrozenBN, GN, "SyncBN", "BN"489_C.MODEL.RESNETS.NORM = "FrozenBN"490 491# Baseline width of each group.492# Scaling this parameters will scale the width of all bottleneck layers.493_C.MODEL.RESNETS.WIDTH_PER_GROUP = 64494 495# Place the stride 2 conv on the 1x1 filter496# Use True only for the original MSRA ResNet; use False for C2 and Torch models497_C.MODEL.RESNETS.STRIDE_IN_1X1 = True498 499# Apply dilation in stage "res5"500_C.MODEL.RESNETS.RES5_DILATION = 1501 502# Output width of res2. Scaling this parameters will scale the width of all 1x1 convs in ResNet503# For R18 and R34, this needs to be set to 64504_C.MODEL.RESNETS.RES2_OUT_CHANNELS = 256505_C.MODEL.RESNETS.STEM_OUT_CHANNELS = 64506 507# Apply Deformable Convolution in stages508# Specify if apply deform_conv on Res2, Res3, Res4, Res5509_C.MODEL.RESNETS.DEFORM_ON_PER_STAGE = [False, False, False, False]510# Use True to use modulated deform_conv (DeformableV2, https://arxiv.org/abs/1811.11168);511# Use False for DeformableV1.512_C.MODEL.RESNETS.DEFORM_MODULATED = False513# Number of groups in deformable conv.514_C.MODEL.RESNETS.DEFORM_NUM_GROUPS = 1515 516 517# ---------------------------------------------------------------------------- #518# Solver519# ---------------------------------------------------------------------------- #520_C.SOLVER = CN()521 522# Options: WarmupMultiStepLR, WarmupCosineLR.523# See detectron2/solver/build.py for definition.524_C.SOLVER.LR_SCHEDULER_NAME = "WarmupMultiStepLR"525 526_C.SOLVER.MAX_ITER = 40000527 528_C.SOLVER.BASE_LR = 0.001529# The end lr, only used by WarmupCosineLR530_C.SOLVER.BASE_LR_END = 0.0531 532_C.SOLVER.MOMENTUM = 0.9533 534_C.SOLVER.NESTEROV = False535 536_C.SOLVER.WEIGHT_DECAY = 0.0001537# The weight decay that's applied to parameters of normalization layers538# (typically the affine transformation)539_C.SOLVER.WEIGHT_DECAY_NORM = 0.0540 541_C.SOLVER.GAMMA = 0.1542# The iteration number to decrease learning rate by GAMMA.543_C.SOLVER.STEPS = (30000,)544# Number of decays in WarmupStepWithFixedGammaLR schedule545_C.SOLVER.NUM_DECAYS = 3546 547_C.SOLVER.WARMUP_FACTOR = 1.0 / 1000548_C.SOLVER.WARMUP_ITERS = 1000549_C.SOLVER.WARMUP_METHOD = "linear"550# Whether to rescale the interval for the learning schedule after warmup551_C.SOLVER.RESCALE_INTERVAL = False552 553# Save a checkpoint after every this number of iterations554_C.SOLVER.CHECKPOINT_PERIOD = 5000555 556# Number of images per batch across all machines. This is also the number557# of training images per step (i.e. per iteration). If we use 16 GPUs558# and IMS_PER_BATCH = 32, each GPU will see 2 images per batch.559# May be adjusted automatically if REFERENCE_WORLD_SIZE is set.560_C.SOLVER.IMS_PER_BATCH = 16561 562# The reference number of workers (GPUs) this config is meant to train with.563# It takes no effect when set to 0.564# With a non-zero value, it will be used by DefaultTrainer to compute a desired565# per-worker batch size, and then scale the other related configs (total batch size,566# learning rate, etc) to match the per-worker batch size.567# See documentation of `DefaultTrainer.auto_scale_workers` for details:568_C.SOLVER.REFERENCE_WORLD_SIZE = 0569 570# Detectron v1 (and previous detection code) used a 2x higher LR and 0 WD for571# biases. This is not useful (at least for recent models). You should avoid572# changing these and they exist only to reproduce Detectron v1 training if573# desired.574_C.SOLVER.BIAS_LR_FACTOR = 1.0575_C.SOLVER.WEIGHT_DECAY_BIAS = None # None means following WEIGHT_DECAY576 577# Gradient clipping578_C.SOLVER.CLIP_GRADIENTS = CN({"ENABLED": False})579# Type of gradient clipping, currently 2 values are supported:580# - "value": the absolute values of elements of each gradients are clipped581# - "norm": the norm of the gradient for each parameter is clipped thus582# affecting all elements in the parameter583_C.SOLVER.CLIP_GRADIENTS.CLIP_TYPE = "value"584# Maximum absolute value used for clipping gradients585_C.SOLVER.CLIP_GRADIENTS.CLIP_VALUE = 1.0586# Floating point number p for L-p norm to be used with the "norm"587# gradient clipping type; for L-inf, please specify .inf588_C.SOLVER.CLIP_GRADIENTS.NORM_TYPE = 2.0589 590# Enable automatic mixed precision for training591# Note that this does not change model's inference behavior.592# To use AMP in inference, run inference under autocast()593_C.SOLVER.AMP = CN({"ENABLED": False})594 595# ---------------------------------------------------------------------------- #596# Specific test options597# ---------------------------------------------------------------------------- #598_C.TEST = CN()599# For end-to-end tests to verify the expected accuracy.600# Each item is [task, metric, value, tolerance]601# e.g.: [['bbox', 'AP', 38.5, 0.2]]602_C.TEST.EXPECTED_RESULTS = []603# The period (in terms of steps) to evaluate the model during training.604# Set to 0 to disable.605_C.TEST.EVAL_PERIOD = 0606# The sigmas used to calculate keypoint OKS. See http://cocodataset.org/#keypoints-eval607# When empty, it will use the defaults in COCO.608# Otherwise it should be a list[float] with the same length as ROI_KEYPOINT_HEAD.NUM_KEYPOINTS.609_C.TEST.KEYPOINT_OKS_SIGMAS = []610# Maximum number of detections to return per image during inference (100 is611# based on the limit established for the COCO dataset).612_C.TEST.DETECTIONS_PER_IMAGE = 100613 614_C.TEST.AUG = CN({"ENABLED": False})615_C.TEST.AUG.MIN_SIZES = (400, 500, 600, 700, 800, 900, 1000, 1100, 1200)616_C.TEST.AUG.MAX_SIZE = 4000617_C.TEST.AUG.FLIP = True618 619_C.TEST.PRECISE_BN = CN({"ENABLED": False})620_C.TEST.PRECISE_BN.NUM_ITER = 200621 622# ---------------------------------------------------------------------------- #623# Misc options624# ---------------------------------------------------------------------------- #625# Directory where output files are written626_C.OUTPUT_DIR = "./output"627# Set seed to negative to fully randomize everything.628# Set seed to positive to use a fixed seed. Note that a fixed seed increases629# reproducibility but does not guarantee fully deterministic behavior.630# Disabling all parallelism further increases reproducibility.631_C.SEED = -1632# Benchmark different cudnn algorithms.633# If input images have very different sizes, this option will have large overhead634# for about 10k iterations. It usually hurts total time, but can benefit for certain models.635# If input images have the same or similar sizes, benchmark is often helpful.636_C.CUDNN_BENCHMARK = False637# The period (in terms of steps) for minibatch visualization at train time.638# Set to 0 to disable.639_C.VIS_PERIOD = 0640 641# global config is for quick hack purposes.642# You can set them in command line or config files,643# and access it with:644#645# from detectron2.config import global_cfg646# print(global_cfg.HACK)647#648# Do not commit any configs into it.649_C.GLOBAL = CN()650_C.GLOBAL.HACK = 1.0651 