Testing the End-to-End Visual Control Model and Fine-Tuning MobileNetV1-SSD with ONNX Export
This week, I finished working on the End-to-End Visual Control exercise by testing the trained model. I also continued improving the fine-tuning of the pre-trained MobileNetV1-SSD model and created a script to export the newly trained model to ONNX for further testing.
Testing End-to-End Visual Control
After exporting the onnx model last week, I tested it in Unibotics by following the instructions provided in the exercise guide and the PyTorch ONNX export tutorial.
After solving some problems related to ROS topic namespaces, I was finally able to test the model successfully and complete the exercise.
End-to-End Visual Control video
Improving the Fine-Tuning of MobileNetV1-SSD and exporting the new model to ONNX
In addition, I continued improving the fine-tuned model by experimenting with different training parameters, particularly different values for the learning rate and base learning rate.
Once I reached a minimum loss value and observed that further adjustments only produced minor improvements, I created a script to export the new model to ONNX. The export script was developed taking into account the essential requirements described in the Visual Object Detection exercise guide.
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
import os
os.environ["CUDA_VISIBLE_DEVICES"] = ""
import torch
import torch.nn as nn
import torchvision
from vision.ssd.mobilenetv1_ssd import create_mobilenetv1_ssd
from vision.ssd.config import mobilenetv1_ssd_config as config
class_names = [
name.strip()
for name in open("models/open-images-model-labels.txt").readlines()
]
person_class_idx = class_names.index("Person")
net = create_mobilenetv1_ssd(
len(class_names),
is_test=True
)
net.load(
"models/mb1-ssd-Epoch-1-Loss-2.8407790957991756.pth"
)
net.to("cpu")
net.eval()
class ONNXExportWrapper(nn.Module):
def __init__(
self,
net,
image_mean,
image_std,
person_class_idx=1,
score_threshold=0.4,
iou_threshold=0.45,
max_detections=20
):
super().__init__()
self.net = net
mean = torch.tensor(
image_mean,
dtype=torch.float32
).view(1, 3, 1, 1)
self.register_buffer("mean", mean)
self.std = float(image_std)
self.person_class_idx = person_class_idx
self.score_threshold = score_threshold
self.iou_threshold = iou_threshold
self.max_detections = max_detections
def forward(self, x):
x = x.permute(0, 3, 1, 2)
x = (x - self.mean) / self.std
confidences, boxes = self.net(x)
confidences = confidences[0]
boxes = boxes[0]
scores = confidences[:, self.person_class_idx]
keep_mask = scores > self.score_threshold
boxes_f = boxes[keep_mask]
scores_f = scores[keep_mask]
keep_idx = torchvision.ops.nms(
boxes_f,
scores_f,
self.iou_threshold
)
keep_idx = keep_idx[:self.max_detections]
final_boxes = boxes_f[keep_idx]
final_scores = scores_f[keep_idx]
x1 = final_boxes[:, 0]
y1 = final_boxes[:, 1]
x2 = final_boxes[:, 2]
y2 = final_boxes[:, 3]
boxes_tlbr = torch.stack(
[y1, x1, y2, x2],
dim=1
)
max_det = self.max_detections
boxes_padding = torch.zeros(
max_det,
4,
dtype=boxes_tlbr.dtype,
device=boxes_tlbr.device
)
scores_padding = torch.zeros(
max_det,
dtype=final_scores.dtype,
device=final_scores.device
)
boxes_tlbr = torch.cat(
[boxes_tlbr, boxes_padding],
dim=0
)[:max_det]
final_scores = torch.cat(
[final_scores, scores_padding],
dim=0
)[:max_det]
detection_boxes = boxes_tlbr.unsqueeze(0)
detection_classes = torch.ones(
1,
max_det,
dtype=torch.float32,
device=boxes_tlbr.device
)
detection_scores = final_scores.unsqueeze(0)
num_detections = torch.minimum(
torch.tensor(
[boxes_tlbr.shape[0]],
dtype=torch.float32,
device=boxes_tlbr.device
),
torch.tensor(
[max_det],
dtype=torch.float32,
device=boxes_tlbr.device
)
)
return (
detection_boxes,
detection_classes,
detection_scores,
num_detections
)
wrapper = ONNXExportWrapper(
net,
image_mean=config.image_mean,
image_std=config.image_std,
person_class_idx=person_class_idx,
score_threshold=0.4,
iou_threshold=0.45,
max_detections=20
)
wrapper.eval()
dummy_input = torch.randn(
1,
300,
300,
3,
dtype=torch.float32
)
torch.onnx.export(
wrapper,
dummy_input,
"models/mb1-ssd-person.onnx",
input_names=[
"input"
],
output_names=[
"detection_boxes",
"detection_classes",
"detection_scores",
"num_detections"
],
opset_version=18
)
print("ONNX model exported successfully.")
I also investigated the pre-trained YOLO model as a possible alternative for fine-tuning and using it in the Visual Object Detection exercise. The objective is to compare both approaches, MobileNetV1-SSD and YOLO, and evaluate their performance in the object detection task.