ssd_vgg16_512.yml 3.2 KB
Newer Older
1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61
architecture: SSD
use_gpu: true
max_iters: 400000
snapshot_iter: 10000
log_smooth_window: 20
log_iter: 20
metric: COCO
pretrain_weights: https://paddle-imagenet-models-name.bj.bcebos.com/VGG16_caffe_pretrained.tar
save_dir: output
weights: output/ssd_vgg16_512/model_final
num_classes: 81

SSD:
  backbone: VGG
  multi_box_head: MultiBoxHead
  output_decoder:
    background_label: 0
    keep_top_k: 200
    nms_eta: 1.0
    nms_threshold: 0.45
    nms_top_k: 400
    score_threshold: 0.01

VGG:
  depth: 16
  with_extra_blocks: true
  normalizations: [20., -1, -1, -1, -1, -1, -1]
  extra_block_filters: [[256, 512, 1, 2, 3], [128, 256, 1, 2, 3], [128, 256, 1, 2, 3], [128, 256, 1, 2, 3], [128, 256, 1, 1, 4]]


MultiBoxHead:
  base_size: 512
  aspect_ratios: [[2.], [2., 3.], [2., 3.], [2., 3.], [2., 3.], [2.], [2.]]
  min_ratio: 15
  max_ratio: 90
  min_sizes: [20.0, 51.0, 133.0, 215.0, 296.0, 378.0, 460.0]
  max_sizes: [51.0, 133.0, 215.0, 296.0, 378.0, 460.0, 542.0]
  steps: [8, 16, 32, 64, 128, 256, 512]
  offset: 0.5
  flip: true
  kernel_size: 3
  pad: 1

LearningRate:
  base_lr: 0.001
  schedulers:
  - !PiecewiseDecay
    gamma: 0.1
    milestones: [280000, 360000]
  - !LinearWarmup
    start_factor: 0.3333333333333333
    steps: 500

OptimizerBuilder:
  optimizer:
    momentum: 0.9
    type: Momentum
  regularizer:
    factor: 0.0005
    type: L2

62 63 64 65
TrainReader:
  inputs_def:
    image_shape: [3, 512, 512]
    fields: ['image', 'gt_bbox', 'gt_class']
66
  dataset:
67 68 69
    !COCODataSet
    image_dir: val2017
    anno_path: annotations/instances_val2017.json
70 71 72 73 74 75 76 77 78
    dataset_dir: dataset/coco
  sample_transforms:
  - !DecodeImage
    to_rgb: true
    with_mixup: false
  - !RandomDistort
    brightness_lower: 0.875
    brightness_upper: 1.125
    is_order: true
79 80 81 82 83
  - !RandomExpand
    fill_value: [104, 117, 123]
  - !RandomCrop
    allow_no_crop: true
  - !NormalizeBox {}
84 85 86 87 88 89 90 91 92 93 94 95 96
  - !ResizeImage
    interp: 1
    target_size: 512
    use_cv2: false
  - !RandomFlipImage
    is_normalized: true
  - !Permute
    to_bgr: false
  - !NormalizeImage
    is_scale: false
    mean: [104, 117, 123]
    std: [1, 1, 1]
  batch_size: 8
97 98
  shuffle: true
  worker_num: 8
99
  bufsize: 16
100
  use_process: true
101 102 103 104

EvalReader:
  inputs_def:
    image_shape: [3,512,512]
105
    fields: ['image', 'gt_bbox', 'gt_class', 'im_shape', 'im_id']
106
  dataset:
107
    !COCODataSet
108
    image_dir: val2017
109 110
    anno_path: annotations/instances_val2017.json
    dataset_dir: dataset/coco
111 112 113 114 115 116 117 118 119 120 121 122 123 124
  sample_transforms:
  - !DecodeImage
    to_rgb: true
    with_mixup: false
  - !ResizeImage
    interp: 1
    target_size: 512
    use_cv2: false
  - !Permute
    to_bgr: false
  - !NormalizeImage
    is_scale: false
    mean: [104, 117, 123]
    std: [1, 1, 1]
125 126
  batch_size: 8
  worker_num: 8
127
  bufsize: 16
128
  drop_empty: false
129

130 131 132 133
TestReader:
  inputs_def:
    image_shape: [3,512,512]
    fields: ['image', 'im_id', 'im_shape']
134
  dataset:
135 136
    !ImageFolder
    anno_path: annotations/instances_val2017.json
137 138 139 140 141 142 143 144
  sample_transforms:
  - !DecodeImage
    to_rgb: true
    with_mixup: false
  - !ResizeImage
    interp: 1
    max_size: 0
    target_size: 512
145
    use_cv2: true
146 147 148 149 150 151
  - !Permute
    to_bgr: false
  - !NormalizeImage
    is_scale: false
    mean: [104, 117, 123]
    std: [1, 1, 1]
152
  batch_size: 1