욜로 나스로 기존의 모델을 이전하기 위한 테스트를 진행
YOLO NAS - SuperGradients
- License : Apache 2.0
- -> 라이센스가 아파치이기 때문에 상업적 이용이 가능
- 꼭 필요한 작업이 될 듯
Model Train
- train_dataset_params 과 valid_dataset_params를 지정
- 학습을 위한 변수설정
- 학습 데이터의 위치와 사이즈, 라벨 파일 등에 대한 설정 과정
from super_gradients.training.datasets.detection_datasets.coco_format_detection import COCOFormatDetectionDataset
from super_gradients.training.transforms.transforms import (
DetectionRandomAffine,
DetectionHSV,
DetectionHorizontalFlip,
DetectionPaddedRescale,
DetectionStandardize,
DetectionTargetsFormatTransform,
)
from super_gradients.training.utils.collate_fn import DetectionCollateFN
train_dataset_params = dict(
data_dir="nas_samsung_640",
images_dir="images/train",
json_annotation_file="train_annotations.coco.json",
input_dim=(640, 640),
ignore_empty_annotations=False,
with_crowd=False,
all_classes_list=CLASS_NAMES,
transforms=[
DetectionRandomAffine(degrees=0.0, scales=(0.5, 1.5), shear=0.0, target_size=(640, 640), filter_box_candidates=False, border_value=128),
DetectionHSV(prob=1.0, hgain=5, vgain=30, sgain=30),
DetectionHorizontalFlip(prob=0.5),
DetectionPaddedRescale(input_dim=(640, 640)),
DetectionStandardize(max_value=255),
DetectionTargetsFormatTransform(input_dim=(640, 640), output_format="LABEL_CXCYWH"),
],
)
valid_dataset_params = dict(
data_dir="nas_samsung_640",
images_dir="images/valid",
json_annotation_file="valid_annotations.coco.json",
input_dim=(640, 640),
ignore_empty_annotations=False,
with_crowd=False,
all_classes_list=CLASS_NAMES,
transforms=[
DetectionPaddedRescale(input_dim=(640, 640), max_targets=300),
DetectionStandardize(max_value=255),
DetectionTargetsFormatTransform(input_dim=(640, 640), output_format="LABEL_CXCYWH"),
],
)
trainset = COCOFormatDetectionDataset(**train_dataset_params)
valset = COCOFormatDetectionDataset(**valid_dataset_params)
- DataLoader 지정
- 배치 사이즈와 worker의 개수 , shuffle의 on off등 학습과 관련된 하이퍼파라미터를 지정하는 과정
- 1번에서 설정한 trainset, valset을 어떤 설정으로 load하여 학습에 사용할지 결정
from torch.utils.data import DataLoader
NUM_WORKERS = 0
BATCH_SIZE = 16
train_dataloader_params = {
"shuffle": True,
"batch_size": BATCH_SIZE,
"drop_last": True,
"pin_memory": True,
"collate_fn": DetectionCollateFN(),
"num_workers": NUM_WORKERS,
"persistent_workers": NUM_WORKERS > 0,
}
val_dataloader_params = {
"shuffle": False,
"batch_size": BATCH_SIZE,
"drop_last": False,
"pin_memory": True,
"collate_fn": DetectionCollateFN(),
"num_workers": NUM_WORKERS,
"persistent_workers": NUM_WORKERS > 0,
}
train_loader = DataLoader(trainset, **train_dataloader_params)
valid_loader = DataLoader(valset, **val_dataloader_params)
- 학습을 위한 learing late, optimizer, epoch 등 학습 방식에 대한 하이퍼파라미터를 지정하는 과정
from super_gradients.training.losses import PPYoloELoss
from super_gradients.training.metrics import DetectionMetrics_050
from super_gradients.training.models.detection_models.pp_yolo_e import PPYoloEPostPredictionCallback
train_params = {
"warmup_initial_lr": 1e-5,
"initial_lr": 5e-4,
"lr_mode": "cosine",
"cosine_final_lr_ratio": 0.5,
"optimizer": "AdamW",
"zero_weight_decay_on_bias_and_bn": True,
"lr_warmup_epochs": 1,
"warmup_mode": "LinearEpochLRWarmup",
"optimizer_params": {"weight_decay": 0.0001},
"ema": False,
"average_best_models": False,
"ema_params": {"beta": 25, "decay_type": "exp"},
"max_epochs": 10,
"mixed_precision": True,
"loss": PPYoloELoss(use_static_assigner=False, num_classes=NUM_CLASSES, reg_max=16),
"valid_metrics_list": [
DetectionMetrics_050(
score_thres=0.1,
top_k_predictions=300,
num_cls=NUM_CLASSES,
normalize_targets=True,
include_classwise_ap=True,
class_names=CLASS_NAMES,
post_prediction_callback=PPYoloEPostPredictionCallback(score_threshold=0.01, nms_top_k=1000, max_predictions=300, nms_threshold=0.7),
)
],
"metric_to_watch": "mAP@0.50",
}
- 학습 실행 코드
from super_gradients.training import Trainer
from super_gradients.common.object_names import Models
from super_gradients.training import models
trainer = Trainer(experiment_name="yolo_nas_s_cppe-5", ckpt_root_dir="experiments")
model = models.get(Models.YOLO_NAS_S, num_classes=NUM_CLASSES, pretrained_weights="coco")
trainer.train(model=model, training_params=train_params, train_loader=train_loader, valid_loader=valid_loader)
Model Inference
- 학습된 모델을 이용해 추론을 진행하는 방법
from super_gradients.common.object_names import Models from super_gradients.training import models NUM_CLASSES = 7
import cv2 import time import numpy as np from ultralytics import YOLO
# 모델 경로
# 비디오 경로 video_path3 = r"C:\Users\edint\Downloads\WIN_20240704_12_51_10_Pro\WIN_20240704_12_48_37_Pro.mp4"
# 모델 로드 # behavior_model = YOLO(behavior_model_path) behavior_model = models.get(Models.YOLO_NAS_S, num_classes=NUM_CLASSES, checkpoint_path=r'C:\SAMSUNG_PROJ\NAS\experiments\yolo_nas_s_cppe-5\RUN_20240809_131001_642791\ckpt_best.pth').cuda()
model = models.get(Models.YOLO_NAS_S, pretrained_weights="coco").cuda()
# 비디오 캡처 설정 cap = cv2.VideoCapture(video_path2) if not cap.isOpened(): print(f"Error: Could not open video file {video_path3}") exit()
# 색상 변수 BLUE = (255, 0, 0) WHITE = (255, 255, 255)
# 모드 변수 behavior_mode = True detection_mode = True
# FPS 변수 prev_time = time.time() fps = 0
while cap.isOpened(): ret, frame = cap.read() org_frame = frame.copy() if not ret: print('Failed to grab frame') break
curr_time = time.time()
detected_people = [] behavior_results_class_names = [] # results = behavior_model.predict(frame, fuse_model=False) results = behavior_model.predict(frame, fuse_model=False)
for i in range (len(results.prediction)): confidence = results.prediction.confidence[i] bbox = results.prediction.bboxes_xyxy[i]
# 박스 좌표를 정수로 변환 xmin = int(bbox[0]) ymin = int(bbox[1]) xmax = int(bbox[2]) ymax = int(bbox[3]) # print(xmin) class_name = results.class_names[results.prediction.labels[i]]
cv2.rectangle(frame, (int(xmin), int(ymin)), (int(xmax), int(ymax)), BLUE, 1) cv2.rectangle(frame, (xmin, ymin - 22), (xmin + 27, ymin), BLUE, -1) cv2.putText(frame, f"{class_name}{int(confidence)}", (xmin + 5, ymin - 8), cv2.FONT_HERSHEY_SIMPLEX, 0.5, WHITE, 2) # FPS 계산 try: fps = 1 / (curr_time - prev_time) prev_time = curr_time # FPS 표시 cv2.putText(frame, f"FPS: {fps:.2f}", (10, 30), cv2.FONT_HERSHEY_SIMPLEX, 1, WHITE, 2) # cv2.putText(frame, f"detected behavior: {len(behavior_results):.2f}", (10, 80), cv2.FONT_HERSHEY_SIMPLEX, 1, WHITE, 2) # cv2.putText(frame, f"filtered behavior: {len(nms_confidences):.2f}", (10, 120), cv2.FONT_HERSHEY_SIMPLEX, 1, WHITE, 2) except ZeroDivisionError: print('ZeroDivisionError: total time is zero.') cv2.imshow("Video", frame)
input_key = cv2.waitKey(1) & 0xFF if input_key == ord("q"): break
cap.release() cv2.destroyAllWindows()
위와 같이 다음 사람이 모델을 이전할 수 있도록 튜토리얼을 만들어 정리해두었다. |