Sairii commited on
Commit
1ee5d04
·
verified ·
1 Parent(s): 6f6e8a1

upload finetuned RT-DETRv2 model on trashify dataset from learn HF course

Browse files
Files changed (5) hide show
  1. README.md +95 -0
  2. config.json +137 -0
  3. model.safetensors +3 -0
  4. preprocessor_config.json +36 -0
  5. training_args.bin +3 -0
README.md ADDED
@@ -0,0 +1,95 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ library_name: transformers
3
+ license: apache-2.0
4
+ base_model: PekingU/rtdetr_v2_r50vd
5
+ tags:
6
+ - generated_from_trainer
7
+ model-index:
8
+ - name: rt_detrv2_finetuned_trashify_box_detector_v1
9
+ results: []
10
+ ---
11
+
12
+ <!-- This model card has been generated automatically according to the information the Trainer had access to. You
13
+ should probably proofread and complete it, then remove this comment. -->
14
+
15
+ # rt_detrv2_finetuned_trashify_box_detector_v1
16
+
17
+ This model is a fine-tuned version of [PekingU/rtdetr_v2_r50vd](https://huggingface.co/PekingU/rtdetr_v2_r50vd) on an unknown dataset.
18
+ It achieves the following results on the evaluation set:
19
+ - Loss: 9.1241
20
+ - Map: 0.43
21
+ - Map 50: 0.5696
22
+ - Map 75: 0.504
23
+ - Map Small: 0.0
24
+ - Map Medium: 0.2939
25
+ - Map Large: 0.4379
26
+ - Mar 1: 0.4831
27
+ - Mar 10: 0.6683
28
+ - Mar 100: 0.7138
29
+ - Mar Small: 0.0
30
+ - Mar Medium: 0.4396
31
+ - Mar Large: 0.7408
32
+ - Map Bin: 0.7822
33
+ - Mar 100 Bin: 0.8782
34
+ - Map Hand: 0.5503
35
+ - Mar 100 Hand: 0.8237
36
+ - Map Not Bin: 0.1143
37
+ - Mar 100 Not Bin: 0.5273
38
+ - Map Not Hand: 0.0108
39
+ - Mar 100 Not Hand: 0.5333
40
+ - Map Not Trash: 0.1591
41
+ - Mar 100 Not Trash: 0.5754
42
+ - Map Trash: 0.6508
43
+ - Mar 100 Trash: 0.8157
44
+ - Map Trash Arm: 0.7426
45
+ - Mar 100 Trash Arm: 0.8429
46
+
47
+ ## Model description
48
+
49
+ More information needed
50
+
51
+ ## Intended uses & limitations
52
+
53
+ More information needed
54
+
55
+ ## Training and evaluation data
56
+
57
+ More information needed
58
+
59
+ ## Training procedure
60
+
61
+ ### Training hyperparameters
62
+
63
+ The following hyperparameters were used during training:
64
+ - learning_rate: 0.0001
65
+ - train_batch_size: 8
66
+ - eval_batch_size: 8
67
+ - seed: 42
68
+ - optimizer: Use OptimizerNames.ADAMW_TORCH_FUSED with betas=(0.9,0.999) and epsilon=1e-08 and optimizer_args=No additional optimizer arguments
69
+ - lr_scheduler_type: linear
70
+ - lr_scheduler_warmup_ratio: 0.05
71
+ - num_epochs: 10
72
+ - mixed_precision_training: Native AMP
73
+
74
+ ### Training results
75
+
76
+ | Training Loss | Epoch | Step | Validation Loss | Map | Map 50 | Map 75 | Map Small | Map Medium | Map Large | Mar 1 | Mar 10 | Mar 100 | Mar Small | Mar Medium | Mar Large | Map Bin | Mar 100 Bin | Map Hand | Mar 100 Hand | Map Not Bin | Mar 100 Not Bin | Map Not Hand | Mar 100 Not Hand | Map Not Trash | Mar 100 Not Trash | Map Trash | Mar 100 Trash | Map Trash Arm | Mar 100 Trash Arm |
77
+ |:-------------:|:-----:|:----:|:---------------:|:------:|:------:|:------:|:---------:|:----------:|:---------:|:------:|:------:|:-------:|:---------:|:----------:|:---------:|:-------:|:-----------:|:--------:|:------------:|:-----------:|:---------------:|:------------:|:----------------:|:-------------:|:-----------------:|:---------:|:-------------:|:-------------:|:-----------------:|
78
+ | 49.9733 | 1.0 | 99 | 12.0835 | 0.3203 | 0.4674 | 0.3424 | 0.0179 | 0.2661 | 0.3322 | 0.34 | 0.5944 | 0.6611 | 0.25 | 0.5244 | 0.6918 | 0.6474 | 0.8574 | 0.5238 | 0.7461 | 0.0675 | 0.6143 | -1.0 | -1.0 | 0.1027 | 0.5667 | 0.5776 | 0.7487 | 0.0031 | 0.4333 |
79
+ | 19.9634 | 2.0 | 198 | 10.2864 | 0.3655 | 0.5109 | 0.4023 | 0.0543 | 0.2008 | 0.3808 | 0.5006 | 0.6538 | 0.6922 | 0.3 | 0.4903 | 0.7206 | 0.7375 | 0.8574 | 0.5285 | 0.7794 | 0.031 | 0.6 | -1.0 | -1.0 | 0.211 | 0.5333 | 0.6379 | 0.7832 | 0.0471 | 0.6 |
80
+ | 16.8253 | 3.0 | 297 | 9.4256 | 0.5134 | 0.6761 | 0.5972 | 0.0409 | 0.2179 | 0.5291 | 0.5318 | 0.7198 | 0.7612 | 0.4 | 0.6131 | 0.7892 | 0.765 | 0.8589 | 0.5846 | 0.7843 | 0.0951 | 0.6357 | -1.0 | -1.0 | 0.1681 | 0.6569 | 0.6675 | 0.7982 | 0.8 | 0.8333 |
81
+ | 15.0193 | 4.0 | 396 | 9.1095 | 0.4935 | 0.6666 | 0.5785 | 0.0308 | 0.3855 | 0.5203 | 0.5489 | 0.7252 | 0.761 | 0.4 | 0.592 | 0.7932 | 0.7817 | 0.8901 | 0.5777 | 0.8098 | 0.1595 | 0.6786 | -1.0 | -1.0 | 0.2067 | 0.6 | 0.6799 | 0.7876 | 0.5552 | 0.8 |
82
+ | 13.5647 | 5.0 | 495 | 8.8113 | 0.5406 | 0.7204 | 0.6351 | 0.001 | 0.2421 | 0.5689 | 0.5632 | 0.7029 | 0.7424 | 0.4 | 0.4847 | 0.7789 | 0.7958 | 0.8851 | 0.6122 | 0.8225 | 0.1892 | 0.5786 | -1.0 | -1.0 | 0.2743 | 0.6014 | 0.681 | 0.8 | 0.6911 | 0.7667 |
83
+ | 12.3311 | 6.0 | 594 | 8.9757 | 0.5597 | 0.7359 | 0.6413 | 0.0021 | 0.2627 | 0.5877 | 0.5873 | 0.7451 | 0.7768 | 0.35 | 0.5716 | 0.8105 | 0.7933 | 0.8865 | 0.5739 | 0.802 | 0.2277 | 0.6571 | -1.0 | -1.0 | 0.2669 | 0.6278 | 0.6625 | 0.7876 | 0.8338 | 0.9 |
84
+ | 11.3095 | 7.0 | 693 | 9.0993 | 0.5256 | 0.7043 | 0.6096 | 0.0003 | 0.2768 | 0.5546 | 0.5767 | 0.7239 | 0.7611 | 0.3 | 0.5909 | 0.7934 | 0.785 | 0.8752 | 0.588 | 0.7902 | 0.1903 | 0.6643 | -1.0 | -1.0 | 0.2666 | 0.6125 | 0.6586 | 0.7912 | 0.6654 | 0.8333 |
85
+ | 10.3959 | 8.0 | 792 | 9.0847 | 0.5307 | 0.7005 | 0.6144 | 0.0 | 0.351 | 0.5598 | 0.5772 | 0.7106 | 0.7491 | 0.0 | 0.5398 | 0.7868 | 0.8053 | 0.8823 | 0.589 | 0.7873 | 0.1919 | 0.6714 | -1.0 | -1.0 | 0.2452 | 0.5694 | 0.6602 | 0.7841 | 0.6923 | 0.8 |
86
+ | 9.7215 | 9.0 | 891 | 9.2169 | 0.5327 | 0.7083 | 0.6145 | 0.0 | 0.2484 | 0.5644 | 0.5807 | 0.7244 | 0.7496 | 0.0 | 0.5193 | 0.7887 | 0.8121 | 0.8823 | 0.5569 | 0.7471 | 0.1826 | 0.6786 | -1.0 | -1.0 | 0.2546 | 0.5694 | 0.657 | 0.7867 | 0.7327 | 0.8333 |
87
+ | 9.1183 | 10.0 | 990 | 9.2996 | 0.5294 | 0.7099 | 0.6115 | 0.0 | 0.2496 | 0.5584 | 0.582 | 0.7183 | 0.7457 | 0.0 | 0.5199 | 0.7829 | 0.7925 | 0.8745 | 0.5496 | 0.7471 | 0.1949 | 0.6643 | -1.0 | -1.0 | 0.2514 | 0.5694 | 0.6555 | 0.7858 | 0.7327 | 0.8333 |
88
+
89
+
90
+ ### Framework versions
91
+
92
+ - Transformers 4.57.1
93
+ - Pytorch 2.9.0+cu126
94
+ - Datasets 4.4.1
95
+ - Tokenizers 0.22.1
config.json ADDED
@@ -0,0 +1,137 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "activation_dropout": 0.0,
3
+ "activation_function": "silu",
4
+ "anchor_image_size": null,
5
+ "architectures": [
6
+ "RTDetrV2ForObjectDetection"
7
+ ],
8
+ "attention_dropout": 0.0,
9
+ "auxiliary_loss": true,
10
+ "backbone": null,
11
+ "backbone_config": {
12
+ "depths": [
13
+ 3,
14
+ 4,
15
+ 6,
16
+ 3
17
+ ],
18
+ "downsample_in_bottleneck": false,
19
+ "downsample_in_first_stage": false,
20
+ "dtype": "float32",
21
+ "embedding_size": 64,
22
+ "hidden_act": "relu",
23
+ "hidden_sizes": [
24
+ 256,
25
+ 512,
26
+ 1024,
27
+ 2048
28
+ ],
29
+ "layer_type": "bottleneck",
30
+ "model_type": "rt_detr_resnet",
31
+ "num_channels": 3,
32
+ "out_features": [
33
+ "stage2",
34
+ "stage3",
35
+ "stage4"
36
+ ],
37
+ "out_indices": [
38
+ 2,
39
+ 3,
40
+ 4
41
+ ],
42
+ "stage_names": [
43
+ "stem",
44
+ "stage1",
45
+ "stage2",
46
+ "stage3",
47
+ "stage4"
48
+ ]
49
+ },
50
+ "backbone_kwargs": null,
51
+ "batch_norm_eps": 1e-05,
52
+ "box_noise_scale": 1.0,
53
+ "d_model": 256,
54
+ "decoder_activation_function": "relu",
55
+ "decoder_attention_heads": 8,
56
+ "decoder_ffn_dim": 1024,
57
+ "decoder_in_channels": [
58
+ 256,
59
+ 256,
60
+ 256
61
+ ],
62
+ "decoder_layers": 6,
63
+ "decoder_method": "default",
64
+ "decoder_n_levels": 3,
65
+ "decoder_n_points": 4,
66
+ "decoder_offset_scale": 0.5,
67
+ "disable_custom_kernels": true,
68
+ "dropout": 0.0,
69
+ "dtype": "float32",
70
+ "encode_proj_layers": [
71
+ 2
72
+ ],
73
+ "encoder_activation_function": "gelu",
74
+ "encoder_attention_heads": 8,
75
+ "encoder_ffn_dim": 1024,
76
+ "encoder_hidden_dim": 256,
77
+ "encoder_in_channels": [
78
+ 512,
79
+ 1024,
80
+ 2048
81
+ ],
82
+ "encoder_layers": 1,
83
+ "eos_coefficient": 0.0001,
84
+ "eval_size": null,
85
+ "feat_strides": [
86
+ 8,
87
+ 16,
88
+ 32
89
+ ],
90
+ "focal_loss_alpha": 0.75,
91
+ "focal_loss_gamma": 2.0,
92
+ "freeze_backbone_batch_norms": true,
93
+ "hidden_expansion": 1.0,
94
+ "id2label": {
95
+ "0": "bin",
96
+ "1": "hand",
97
+ "2": "not_bin",
98
+ "3": "not_hand",
99
+ "4": "not_trash",
100
+ "5": "trash",
101
+ "6": "trash_arm"
102
+ },
103
+ "initializer_bias_prior_prob": null,
104
+ "initializer_range": 0.01,
105
+ "is_encoder_decoder": true,
106
+ "label2id": {
107
+ "bin": 0,
108
+ "hand": 1,
109
+ "not_bin": 2,
110
+ "not_hand": 3,
111
+ "not_trash": 4,
112
+ "trash": 5,
113
+ "trash_arm": 6
114
+ },
115
+ "label_noise_ratio": 0.5,
116
+ "layer_norm_eps": 1e-05,
117
+ "learn_initial_query": false,
118
+ "matcher_alpha": 0.25,
119
+ "matcher_bbox_cost": 5.0,
120
+ "matcher_class_cost": 2.0,
121
+ "matcher_gamma": 2.0,
122
+ "matcher_giou_cost": 2.0,
123
+ "model_type": "rt_detr_v2",
124
+ "normalize_before": false,
125
+ "num_denoising": 100,
126
+ "num_feature_levels": 3,
127
+ "num_queries": 300,
128
+ "positional_encoding_temperature": 10000,
129
+ "transformers_version": "4.57.1",
130
+ "use_focal_loss": true,
131
+ "use_pretrained_backbone": false,
132
+ "use_timm_backbone": false,
133
+ "weight_loss_bbox": 5.0,
134
+ "weight_loss_giou": 2.0,
135
+ "weight_loss_vfl": 1.0,
136
+ "with_box_refine": true
137
+ }
model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a60589006da8081a20e2ec35bc3edb08329fa04f6a392cb17493f6a1ca8ef8ae
3
+ size 171576780
preprocessor_config.json ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "crop_size": null,
3
+ "data_format": "channels_first",
4
+ "default_to_square": false,
5
+ "device": null,
6
+ "disable_grouping": null,
7
+ "do_center_crop": null,
8
+ "do_convert_annotations": true,
9
+ "do_convert_rgb": null,
10
+ "do_normalize": false,
11
+ "do_pad": true,
12
+ "do_rescale": true,
13
+ "do_resize": true,
14
+ "format": "coco_detection",
15
+ "image_mean": [
16
+ 0.485,
17
+ 0.456,
18
+ 0.406
19
+ ],
20
+ "image_processor_type": "RTDetrImageProcessorFast",
21
+ "image_std": [
22
+ 0.229,
23
+ 0.224,
24
+ 0.225
25
+ ],
26
+ "input_data_format": null,
27
+ "pad_size": null,
28
+ "resample": 2,
29
+ "rescale_factor": 0.00392156862745098,
30
+ "return_segmentation_masks": null,
31
+ "return_tensors": null,
32
+ "size": {
33
+ "longest_edge": 640,
34
+ "shortest_edge": 640
35
+ }
36
+ }
training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:089cb1c148dccf4f879f1540aad5fc5f67d32bb8a6cb251eeb46940d5c9f397a
3
+ size 5905