diff --git a/.gitignore b/.gitignore index 01863d0..eae859f 100644 --- a/.gitignore +++ b/.gitignore @@ -1,6 +1,6 @@ api/data/ api/saved_models/ api/testimages/ -api/__pycache__/ +**/__pycache__/ **/*.env \ No newline at end of file diff --git a/api/CocoDetectionDataset.py b/api/CocoDetectionDataset.py deleted file mode 100644 index 6f27871..0000000 --- a/api/CocoDetectionDataset.py +++ /dev/null @@ -1,65 +0,0 @@ -# Custom PyTorch Dataset to load COCO-format annotations and images -import matplotlib.pyplot as plt -import cv2 -import os -import torch -from PIL import Image -import numpy as np -from torch.utils.data import Dataset -from pycocotools.coco import COCO - -# website: https://visionbrick.com/pipeline-for-training-custom-faster-rcnn-object-detection-models-with-pytorch/ - -class CocoDetectionDataset(Dataset): - # Init function: loads annotation file and prepares list of image IDs - def __init__(self, image_dir, annotation_path, transforms=None): - self.image_dir = image_dir - self.coco = COCO(annotation_path) - self.image_ids = list(self.coco.imgs.keys()) - self.transforms = transforms - - # Returns total number of images - def __len__(self): - return len(self.image_ids) - - # Fetches a single image and its annotations - def __getitem__(self, idx): - image_id = self.image_ids[idx] - image_info = self.coco.loadImgs(image_id)[0] - image_path = os.path.join(self.image_dir, image_info['file_name']) - image = Image.open(image_path).convert("RGB") - - # Load all annotations for this image - annotation_ids = self.coco.getAnnIds(imgIds=image_id) - annotations = self.coco.loadAnns(annotation_ids) - - # Extract bounding boxes and labels from annotations - boxes = [] - labels = [] - for obj in annotations: - xmin, ymin, width, height = obj['bbox'] - xmax = xmin + width - ymax = ymin + height - boxes.append([xmin, ymin, xmax, ymax]) - labels.append(obj['category_id']) - - # Convert annotations to PyTorch tensors - boxes = torch.as_tensor(boxes, dtype=torch.float32) - labels = torch.as_tensor(labels, dtype=torch.int64) - area = torch.as_tensor([obj['area'] for obj in annotations], dtype=torch.float32) - iscrowd = torch.as_tensor([obj.get('iscrowd', 0) for obj in annotations], dtype=torch.int64) - - # Package everything into a target dictionary - target = { - "boxes": boxes, - "labels": labels, - "image_id": image_id, - "area": area, - "iscrowd": iscrowd - } - - # Apply transforms if any were passed - if self.transforms: - image = self.transforms(image) - - return image, target \ No newline at end of file diff --git a/api/r-cnn.py b/api/r-cnn.py deleted file mode 100644 index 22792a1..0000000 --- a/api/r-cnn.py +++ /dev/null @@ -1,51 +0,0 @@ -import torch -import torch.nn as nn -import torchvision -from torchvision import datasets, transforms -from torch.utils.data import DataLoader, random_split -import matplotlib.pyplot as plt - -from bird_cnn import Bird_CNN, sample, trainCNN - -from enum import Enum - -from PIL import Image - -SAVE_PATH = "./saved_models" -IMAGE_SIZE = 64 - -class bird_species(Enum): - Common_Kingfisher = 0 - CommonMyna = 1 - House_Crow = 2 - Indian_Peacock = 3 - Indian_Pitta = 4 - Ruddy_Shelduck = 5 - Sarus_Crane = 6 - -transform = transforms.Compose([ - transforms.Resize(IMAGE_SIZE), - transforms.CenterCrop(IMAGE_SIZE), - transforms.ToTensor() -]) - -dataset = datasets.ImageFolder("data/train", transform=transform) - -train_size = int(0.8 * len(dataset)) -val_size = len(dataset) - train_size - -train_dataset, val_dataset = random_split(dataset, [train_size, val_size]) - -train_loader = DataLoader(train_dataset, batch_size=32, shuffle=True) -val_loader = DataLoader(val_dataset, batch_size=32, shuffle=False) - - -device = torch.device("cpu") if not torch.cuda.is_available() else torch.device("cuda:0") -print("Using device", device) - -model = Bird_CNN(c_in=3, c_hidden=32, c_out=7) -model.to(device) -optimizer = torch.optim.Adam(model.parameters(), lr=1e-3, weight_decay=1e-4) -loss_module = nn.CrossEntropyLoss() - -trainCNN(model, optimizer, loss_module, train_loader, val_loader, device, 50, SAVE_PATH=SAVE_PATH, save=True) \ No newline at end of file diff --git a/api/selective_search.py b/api/selective_search.py deleted file mode 100644 index a6339e4..0000000 --- a/api/selective_search.py +++ /dev/null @@ -1,85 +0,0 @@ -from collections import defaultdict -from enum import Enum -import os - -import numpy as np -import matplotlib.pyplot as plt -import cv2 - -import torch -from torchvision import transforms -import torch.nn.functional as F - -from bird_cnn import Bird_CNN - -from PIL import Image - - - -BUILD_PATH = "./build_models" -IMAGE_SIZE = 64 - -class bird_species(Enum): - Common_Kingfisher = 0 - CommonMyna = 1 - House_Crow = 2 - Indian_Peacock = 3 - Indian_Pitta = 4 - Ruddy_Shelduck = 5 - Sarus_Crane = 6 - -transform = transforms.Compose([ - transforms.Resize(IMAGE_SIZE), - transforms.CenterCrop(IMAGE_SIZE), - transforms.ToTensor() -]) - -device = torch.device("cuda" if torch.cuda.is_available() else "cpu") - -model = Bird_CNN(c_in=3, c_hidden=16, c_out=7) -full_path = os.path.join(BUILD_PATH, "bird_cnn") -model.load_state_dict(torch.load(full_path, map_location=torch.device(device))) -model.to(device) -model.eval() - -img = cv2.imread("./testimages/two_crows.jpg") - -if img is None: - raise ValueError("Image not found or path is wrong") - -ss = cv2.ximgproc.segmentation.createSelectiveSearchSegmentation() -ss.setBaseImage(img) - -ss.switchToSelectiveSearchFast() -rects = ss.process() - -# convert to array for easy sorting -rects = np.array(rects) - -# compute area -areas = rects[:, 2] * rects[:, 3] - -# sort by area (descending) -idx = np.argsort(-areas) - -# take top 10 -top10 = rects[idx[:200]] -class_conf_sum = defaultdict(float) - -for (x, y, w, h) in top10: - #cv2.rectangle(img_copy, (x, y), (x + w, y + h), (0, 255, 0), 1) - crop = img[y:y+h, x:x+w] - crop_pil = Image.fromarray(cv2.cvtColor(crop, cv2.COLOR_BGR2RGB)) - image = transform(crop_pil).unsqueeze(0).to(device) - - with torch.no_grad(): - pred = model(image) - probs = F.softmax(pred, dim=1) - confidence, cls = torch.max(probs, dim=1) - if confidence.item() < 0.7: - continue - - class_conf_sum[cls.item()] += confidence.item() - -for i in range(6): - print(str(i) + ": " + str(class_conf_sum[i])) diff --git a/api/bird_cnn.py b/api/src/bird_cnn/bird_cnn.py similarity index 99% rename from api/bird_cnn.py rename to api/src/bird_cnn/bird_cnn.py index 9b2d3ee..1873c9f 100644 --- a/api/bird_cnn.py +++ b/api/src/bird_cnn/bird_cnn.py @@ -1,5 +1,4 @@ import os - import torch import torch.nn as nn import torch.nn.functional as F diff --git a/api/train_bird.py b/api/src/bird_cnn/train_bird.py similarity index 96% rename from api/train_bird.py rename to api/src/bird_cnn/train_bird.py index a7cf9fe..829d6cd 100644 --- a/api/train_bird.py +++ b/api/src/bird_cnn/train_bird.py @@ -6,13 +6,13 @@ from torch.utils.data import DataLoader, random_split import matplotlib.pyplot as plt from util import TransformedSubset, visualizeData -from bird_cnn import Bird_CNN, sample, trainCNN +from .bird_cnn import Bird_CNN, sample, trainCNN from enum import Enum from PIL import Image -SAVE_PATH = "./saved_models" +SAVE_PATH = "../saved_models" IMAGE_SIZE = 128 class bird_species(Enum): diff --git a/api/src/r_cnn/cocoDetectionDataset.py b/api/src/r_cnn/cocoDetectionDataset.py new file mode 100644 index 0000000..eb4b3ab --- /dev/null +++ b/api/src/r_cnn/cocoDetectionDataset.py @@ -0,0 +1,125 @@ +import os +import torch +from PIL import Image +from torch.utils.data import Dataset +from pycocotools.coco import COCO + +class CocoDetectionDataset(Dataset): + # Init function: loads annotation file and prepares list of image IDs + def __init__(self, image_dir, annotation_path, transforms=None): + self.image_dir = image_dir + self.coco = COCO(annotation_path) + self.image_ids = list(self.coco.imgs.keys()) + self.transforms = transforms + + # Returns total number of images + def __len__(self): + return len(self.image_ids) + + # Fetches a single image and its annotations + def __getitem__(self, idx): + image_id = self.image_ids[idx] + image_info = self.coco.loadImgs(image_id)[0] + image_path = os.path.join(self.image_dir, image_info['file_name']) + image = Image.open(image_path).convert("RGB") + + # Load all annotations for this image + annotation_ids = self.coco.getAnnIds(imgIds=image_id) + annotations = self.coco.loadAnns(annotation_ids) + + # Extract bounding boxes and labels from annotations + boxes = [] + labels = [] + for obj in annotations: + xmin, ymin, width, height = obj['bbox'] + xmin, ymin, width, height = float(xmin), float(ymin), float(width), float(height) + xmax = xmin + width + ymax = ymin + height + boxes.append([xmin, ymin, xmax, ymax]) + labels.append(obj['category_id']) + + # Convert annotations to PyTorch tensors + boxes = torch.as_tensor(boxes, dtype=torch.float32) + labels = torch.as_tensor(labels, dtype=torch.int64) + area = torch.as_tensor([obj['area'] for obj in annotations], dtype=torch.float32) + iscrowd = torch.as_tensor([obj.get('iscrowd', 0) for obj in annotations], dtype=torch.int64) + + # Package everything into a target dictionary + target = { + "boxes": boxes, + "labels": labels, + "image_id": image_id, + "area": area, + "iscrowd": iscrowd + } + + # Apply transforms if any were passed + if self.transforms: + image = self.transforms(image) + + return image, target + + +class CocoDetectionDatasetResized(Dataset): + def __init__(self, image_dir, annotation_path, img_size=64, transforms=None): + self.image_dir = image_dir + self.coco = COCO(annotation_path) + self.image_ids = list(self.coco.imgs.keys()) + self.transforms = transforms + self.img_size = img_size + + def __len__(self): + return len(self.image_ids) + + def __getitem__(self, idx): + image_id = self.image_ids[idx] + image_info = self.coco.loadImgs(image_id)[0] + image_path = os.path.join(self.image_dir, image_info['file_name']) + + image = Image.open(image_path).convert("RGB") + + orig_w, orig_h = image.size + + scale_w = self.img_size / orig_w + scale_h = self.img_size / orig_h + + # Load annotations + annotation_ids = self.coco.getAnnIds(imgIds=image_id) + annotations = self.coco.loadAnns(annotation_ids) + + boxes = [] + labels = [] + + for obj in annotations: + xmin, ymin, width, height = obj['bbox'] + xmin, ymin, width, height = float(xmin), float(ymin), float(width), float(height) + + xmin = xmin * scale_w + ymin = ymin * scale_h + xmax = (xmin + width * scale_w) + ymax = (ymin + height * scale_h) + + boxes.append([xmin, ymin, xmax, ymax]) + labels.append(obj['category_id']) + + boxes = torch.tensor(boxes, dtype=torch.float32) + labels = torch.tensor(labels, dtype=torch.long) + + area = torch.tensor([obj['area'] * scale_w * scale_h for obj in annotations], dtype=torch.float32) + iscrowd = torch.tensor([obj.get('iscrowd', 0) for obj in annotations], dtype=torch.long) + + target = { + "boxes": boxes, + "labels": labels, + "image_id": torch.tensor([image_id]), + "area": area, + "iscrowd": iscrowd + } + + # resize image + image = image.resize((self.img_size, self.img_size)) + + if self.transforms: + image = self.transforms(image) + + return image, target \ No newline at end of file diff --git a/api/src/r_cnn/createObjectDetectionDataset.py b/api/src/r_cnn/createObjectDetectionDataset.py new file mode 100644 index 0000000..bf3cc4b --- /dev/null +++ b/api/src/r_cnn/createObjectDetectionDataset.py @@ -0,0 +1,118 @@ +from pathlib import Path +from tqdm import tqdm +from ..util import iou +from .cocoDetectionDataset import CocoDetectionDataset +from torchvision.transforms import ToPILImage, ToTensor +import torch +import torchvision.transforms.functional as TF +from torch.utils.data import DataLoader +import random + + + +SAVE_PATH = "./saved_models" + +def get_transform(): + return ToTensor() + +dataset = CocoDetectionDataset( + image_dir="data/faces/train", + annotation_path="data/faces/train/_annotations.coco.json", + transforms=get_transform() +) + +processing_loader = DataLoader(dataset, batch_size=1, shuffle=False, collate_fn=lambda x: tuple(zip(*x))) + +def random_crop(W, H, sizeX=64, sizeY =64): + x = random.randint(0, max(W - sizeX, 0)) + y = random.randint(0, max(H - sizeY, 0)) + return x, y, x + sizeX, y + sizeY + +def box_inside(inner_box, outer_box): + return ( + inner_box[0] >= outer_box[0] and + inner_box[1] >= outer_box[1] and + inner_box[2] <= outer_box[2] and + inner_box[3] <= outer_box[3] + ) + +def is_background(crop, gt_boxes, threshold=0.0): + for box in gt_boxes: + if iou(crop, box) > threshold: + return False + return True + +def get_background_crop(image, gt_boxes, max_trials=100): + C, H, W = image.shape + + for _ in range(max_trials): + x1, y1, x2, y2 = random_crop(W, H, sizeX=random.randint(50, 600), sizeY=random.randint(50, 600)) + + crop_box = (x1, y1, x2, y2) + + if is_background(crop_box, gt_boxes, threshold=0.0): + crop = image[:, y1:y2, x1:x2] + return crop + + return None + +def create_background_tensor(amount, dataset, labels, boxes, image): + for _ in range(amount): + background = get_background_crop(image=image, gt_boxes=boxes.to(torch.int64)) + if background is None: + break + background = TF.resize(background, [64, 64], antialias=True) + dataset.append(background) + labels.append(torch.tensor(0)) + return dataset, labels + +def create_stack(images, annotations, PADDING): + croped_images = [] + labels = [] + for i in range(len(images)): + image = images[i] + annotation = annotations[i] + _, h, w = image.shape + + for box, label in tuple(zip(annotation["boxes"], annotation["labels"])): + box = box.to(torch.int64) + box_copy = [] + + box_copy.append(max(box[0]-PADDING, 0)) + box_copy.append(max(box[1]-PADDING, 0)) + box_copy.append(min(box[2]+PADDING, w-1)) + box_copy.append(min(box[3]+PADDING, h-1)) + + if box_copy[1] == box_copy[3] or box_copy[0] == box_copy[2]: + continue + + croped_image = image[: , box_copy[1]:box_copy[3], box_copy[0]:box_copy[2]] + croped_image = TF.resize(croped_image, [64, 64], antialias=True) + croped_images.append(croped_image) + labels.append(label) + + croped_images, labels = create_background_tensor(amount=10, dataset=croped_images, labels=labels, boxes=annotation["boxes"], image=image) + return croped_images, labels + +def create_croped_dataset(): + to_pil = ToPILImage() + + out_path = Path("data/faces_processed/face") + out_path.mkdir(parents=True, exist_ok=True) + out_path = Path("data/faces_processed/background") + out_path.mkdir(parents=True, exist_ok=True) + + count_image = 0 + count_background = 0 + for images, annotations in tqdm(processing_loader): + croped_images, labels = create_stack(images, annotations, PADDING=0) + + for i in range(len(croped_images)): + pil_img = to_pil(croped_images[i]) + + if labels[i] == 1: + pil_img.save(f"data/faces_processed/face/{count_image}.jpg") + count_image += 1 + else: + pil_img.save(f"data/faces_processed/background/{count_background}.jpg") + count_background += 1 \ No newline at end of file diff --git a/api/src/r_cnn/r_cnn.py b/api/src/r_cnn/r_cnn.py new file mode 100644 index 0000000..73a4c03 --- /dev/null +++ b/api/src/r_cnn/r_cnn.py @@ -0,0 +1,313 @@ +import os +import cv2 +import torch +import torchvision.transforms.functional as TF +from tqdm import tqdm +import torch.nn as nn +from src.util import iou, visualizeImage +import random +import torch.nn.functional as F + + +class ObjectDetectionCNN(nn.Module): + def __init__(self, c_in, c_hidden, c_out, layers): + super().__init__() + + self.model = nn.ModuleList() + + self.model.append(nn.Sequential( + nn.Conv2d(c_in, c_hidden, kernel_size=3, padding=1), + nn.BatchNorm2d(c_hidden), + nn.ReLU(inplace=True) + )) + + for _ in range(layers-1): + self.model.append(nn.Sequential( + nn.Conv2d(c_hidden, c_hidden, kernel_size=3, padding=1), + nn.BatchNorm2d(c_hidden), + nn.ReLU(inplace=True) + )) + + self.model.append(nn.Sequential( + nn.AdaptiveAvgPool2d((1, 1)), + nn.Flatten(), + nn.Linear(c_hidden, c_out), + nn.Dropout(0.3) + )) + + def forward(self, x): + for layer in self.model: + x = layer(x) + return x + + + +def random_crop(W, H, sizeX=64, sizeY =64): + x = random.randint(0, max(W - sizeX, 0)) + y = random.randint(0, max(H - sizeY, 0)) + return x, y, x + sizeX, y + sizeY + +def is_background(crop, gt_boxes, threshold=0.0): + for box in gt_boxes: + if iou(crop, box) > threshold: + return False + return True + +def get_background_crop(image, gt_boxes, max_trials=100): + C, H, W = image.shape + + for _ in range(max_trials): + x1, y1, x2, y2 = random_crop(W, H, sizeX=random.randint(50, 600), sizeY=random.randint(50, 600)) + + crop_box = (x1, y1, x2, y2) + + if is_background(crop_box, gt_boxes, threshold=0.1): + crop = image[:, y1:y2, x1:x2] + return crop + + return None + +def create_background_tensor(amount, dataset, labels, boxes, image): + for _ in range(amount): + background = get_background_crop(image=image, gt_boxes=boxes.to(torch.int64)) + if background is not None: + background = TF.resize(background, [64, 64], antialias=True) + dataset.append(background) + labels.append(torch.tensor(0)) + else: + print("Background not found") + return dataset, labels + +def create_stack(images, annotations, PADDING): + croped_images = [] + labels = [] + for i in range(len(images)): + image = images[i] + annotation = annotations[i] + _, h, w = image.shape + + for box, label in tuple(zip(annotation["boxes"], annotation["labels"])): + box = box.to(torch.int64) + box_copy = [] + + box_copy.append(max(box[0]-PADDING, 0)) + box_copy.append(max(box[1]-PADDING, 0)) + box_copy.append(min(box[2]+PADDING, w-1)) + box_copy.append(min(box[3]+PADDING, h-1)) + + + croped_image = image[: , box_copy[1]:box_copy[3], box_copy[0]:box_copy[2]] + #visualizeImage(croped_image) + croped_image = TF.resize(croped_image, [64, 64], antialias=True) + #visualizeImage(croped_image) + croped_images.append(croped_image) + labels.append(label) + + #croped_images, labels = create_background_tensor(amount=len(croped_images), dataset=croped_images, labels=labels, boxes=annotation["boxes"], image=image) + return croped_images, labels + +def train(model, loss_module, train_loader, val_loader, optimizer, SAVE_PATH, model_name, saving=True, PADDING=20, device="cpu"): + best_val = torch.finfo(torch.float32).max + + for epoch in range(200): + ############ + # Training # + ############ + model.train() + + true_preds, count, lossCount = 0, 0, 0. + for images, annotations in tqdm(train_loader, desc=f"Train", leave=False): + croped_images, labels = create_stack(images, annotations, PADDING) + + croped_images = torch.stack(croped_images).to(device) + labels = torch.stack(labels).to(device) + + prediction = model(croped_images) + + loss = loss_module(prediction, labels) + lossCount += loss.sum().item() + + optimizer.zero_grad() + + loss.backward() + + optimizer.step() + + true_preds += (prediction.argmax(dim=1) == labels).sum().item() + count += croped_images.size(0) + + train_acc = true_preds / count + train_loss = lossCount / count + + torch.cuda.empty_cache() + + ############## + # Validation # + ############## + model.eval() + + true_preds, count, lossCount = 0, 0, 0. + for images, annotations in tqdm(val_loader, desc=f"Test", leave=False): + with torch.no_grad(): + croped_images, labels = create_stack(images, annotations, PADDING) + + croped_images = torch.stack(croped_images).to(device) + labels = torch.stack(labels).to(device) + + prediction = model(croped_images) + + loss = loss_module(prediction, labels) + lossCount += loss.sum().item() + + true_preds += (prediction.argmax(dim=1) == labels).sum().item() + count += croped_images.size(0) + + val_acc = true_preds / count + val_loss = lossCount / count + + if(saving and best_val > val_loss): + best_val = val_loss + save_dir = os.path.join(SAVE_PATH, model_name) + os.makedirs(save_dir, exist_ok=True) + + save_path = os.path.join(save_dir, model_name) + torch.save(model.state_dict(), save_path) + + print(f"epoch: {epoch+1} | train accuracy: {int(train_acc * 1000) / 10}% | validation accuracy: {int(val_acc * 1000) / 10}% | train loss: {int(train_loss * 1000) / 100} | val loss: {int(val_loss * 1000) / 100}") + torch.cuda.empty_cache() + return best_val + + +def trainNormalDataset(model, loss_module, train_loader, val_loader, optimizer, SAVE_PATH, model_name, saving=True, device="cpu"): + best_val = torch.finfo(torch.float32).max + + for epoch in range(200): + ############ + # Training # + ############ + model.train() + + true_preds, count, lossCount = 0, 0, 0. + for images, labels in tqdm(train_loader, desc=f"Train", leave=False): + images = images.to(device) + labels = labels.to(device) + + prediction = model(images) + + loss = loss_module(prediction, labels) + lossCount += loss.sum().item() + + optimizer.zero_grad() + + loss.backward() + + optimizer.step() + + true_preds += (prediction.argmax(dim=1) == labels).sum().item() + count += images.size(0) + + train_acc = true_preds / count + train_loss = lossCount / count + + torch.cuda.empty_cache() + + ############## + # Validation # + ############## + model.eval() + + true_preds, count, lossCount = 0, 0, 0. + for images, labels in tqdm(val_loader, desc=f"Test", leave=False): + with torch.no_grad(): + images = images.to(device) + labels = labels.to(device) + + prediction = model(images) + + loss = loss_module(prediction, labels) + lossCount += loss.sum().item() + + true_preds += (prediction.argmax(dim=1) == labels).sum().item() + count += images.size(0) + + val_acc = true_preds / count + val_loss = lossCount / count + + if(saving and best_val > val_loss): + best_val = val_loss + save_dir = os.path.join(SAVE_PATH, model_name) + os.makedirs(save_dir, exist_ok=True) + + save_path = os.path.join(save_dir, model_name) + torch.save(model.state_dict(), save_path) + + print(f"epoch: {epoch+1} | train accuracy: {int(train_acc * 1000) / 10}% | validation accuracy: {int(val_acc * 1000) / 10}% | train loss: {int(train_loss * 1000) / 100} | val loss: {int(val_loss * 1000) / 100}") + torch.cuda.empty_cache() + return best_val + +def resize_keep_aspect(img, target_w, target_h): + h, w = img.shape[:2] + + scale = min(target_w / w, target_h / h) + new_w = int(w * scale) + new_h = int(h * scale) + + resized = cv2.resize(img, (new_w, new_h), interpolation=cv2.INTER_AREA) + return resized, w / new_w, h / new_h + +def map_box_to_original(box, scale_x, scale_y): + x1, y1, x2, y2 = box + + return int(x1 * scale_x), int(y1 * scale_y), int(x2 * scale_x), int(y2 * scale_y) + +def eval(model, image, BUILD_PATH, device, PADDING = 0, minSize=5, maxSize=600, minConf=0.8): + _, H, W = image.shape + + model.load_state_dict(torch.load(BUILD_PATH, map_location=torch.device(device))) + model.to(device) + model.eval() + image_numpy = image.permute(1, 2, 0).cpu().numpy() + image_numpy = (image_numpy*255).astype("uint8") + image_numpy = cv2.cvtColor(image_numpy, cv2.COLOR_BGR2RGB) + image_numpy_copy, resizedW, resizedH = resize_keep_aspect(image_numpy, W, H) + + cv2.imshow("", image_numpy_copy) + cv2.waitKey(0) + + ss = cv2.ximgproc.segmentation.createSelectiveSearchSegmentation() + ss.setBaseImage(image_numpy_copy) + ss.switchToSelectiveSearchFast() + + rects = ss.process() + + draw = image_numpy.copy() + predictions = [] + + for (x, y, w, h) in tqdm(rects): + x, y, w, h = map_box_to_original((x, y, w, h), resizedW, resizedH) + + if w < minSize or h < minSize or w > maxSize or h > maxSize: + continue + + x1 = max(x-PADDING, 0) + y1 = max(y-PADDING, 0) + x2 = min(x+w+PADDING, W-1) + y2 = min(y+h+PADDING, H-1) + + crop = image[:, y1:y2, x1:x2] + crop = TF.resize(crop, [64, 64], antialias=True) + crop = crop.unsqueeze(0).to(device) + + pred = model(crop) + probs = F.softmax(pred, dim=1) + confidence, cls = torch.max(probs, dim=1) + if cls.item() == 1 and confidence.item() > minConf: + predictions.append((confidence.item(), cls.item(), (x, y, w, h))) + + predictions = sorted(predictions, key=lambda x: x[0], reverse=True) + print(len(predictions)) + for conf, cls, (x, y, w, h) in predictions[:]: + cv2.rectangle(draw, (x, y), (x + w, y + h), (255, 0, 0), 1) + + cv2.imshow("", draw) + cv2.waitKey(0) \ No newline at end of file diff --git a/api/src/r_cnn/r_cnn_test.py b/api/src/r_cnn/r_cnn_test.py new file mode 100644 index 0000000..7f0b4b9 --- /dev/null +++ b/api/src/r_cnn/r_cnn_test.py @@ -0,0 +1,56 @@ +import json +import os +import cv2 +from matplotlib import pyplot as plt +import numpy as np +from torch import tensor +import torch +from torch.utils.data import DataLoader +from torchvision import transforms +from torchvision.transforms import ToTensor, functional +import torchvision.transforms.functional as TF +from tqdm import tqdm +import torch.nn as nn +import torch.nn.functional as F + +from .r_cnn import ObjectDetectionCNN, train, eval +from .cocoDetectionDataset import CocoDetectionDataset +import random + +SAVE_PATH = "./saved_models" +PADDING = 20 + +def get_transform(): + return ToTensor() + +train_dataset = CocoDetectionDataset( + image_dir="data/football/train", + annotation_path="data/football/train/_annotations.coco.json", + transforms=get_transform() +) + +val_dataset = CocoDetectionDataset( + image_dir="data/football/valid", + annotation_path="data/football/valid/_annotations.coco.json", + transforms=get_transform() +) + +train_loader = DataLoader(train_dataset, batch_size=2, shuffle=True, collate_fn=lambda x: tuple(zip(*x))) +val_loader = DataLoader(val_dataset, batch_size=2, shuffle=True, collate_fn=lambda x: tuple(zip(*x))) + +device = torch.device("cpu") if not torch.cuda.is_available() else torch.device("cuda:0") +print("Using device", device) + +model = ObjectDetectionCNN(c_in=3, c_hidden=32, c_out=2, layers=10) +model.to(device) +optimizer = torch.optim.Adam(model.parameters(), lr=1e-3, weight_decay=1e-4) +loss_module = nn.CrossEntropyLoss() + +#train(model=model, loss_module=loss_module, train_loader=train_loader, val_loader=val_loader, +# optimizer=optimizer, SAVE_PATH=SAVE_PATH, saving=True, PADDING=40, device=device) + +#exit() + +image = next(iter(val_loader))[0][0] +eval(model=model, image=image, BUILD_PATH=os.path.join(SAVE_PATH, "object_detection", "object_detection"), + device=device, PADDING=40, minSize=5, maxSize=100, minConf=0.8) \ No newline at end of file diff --git a/api/src/r_cnn/train_face_detection_r_cnn.py b/api/src/r_cnn/train_face_detection_r_cnn.py new file mode 100644 index 0000000..e8f545e --- /dev/null +++ b/api/src/r_cnn/train_face_detection_r_cnn.py @@ -0,0 +1,72 @@ +import os + +from torchvision import transforms +from tqdm import tqdm +from util import TransformedSubset, visualizeImage +from api.src.r_cnn.r_cnn import ObjectDetectionCNN, trainNormalDataset, eval +from torch.utils.data import DataLoader, random_split +from torchvision import datasets +import torch +import torch.nn as nn +from torchvision.io import read_image +from torch.utils.data import WeightedRandomSampler + + + +SAVE_PATH = "./saved_models" +PADDING = 20 +IMAGE_SIZE = 64 + +transform = transforms.Compose([ + transforms.Resize(IMAGE_SIZE), + transforms.CenterCrop(IMAGE_SIZE), + transforms.ToTensor() +]) + +transform_augemnt = transforms.Compose([ + transforms.RandomHorizontalFlip(p=0.5), + transforms.RandomRotation(degrees=15), + transforms.Resize(IMAGE_SIZE), + transforms.CenterCrop(IMAGE_SIZE), + transforms.ToTensor() +]) + +dataset = datasets.ImageFolder("data/faces_processed") + +weights = [1/3, 1] + +sampler = WeightedRandomSampler( + weights=weights, + num_samples=len(dataset), + replacement=True +) + +train_size = int(0.8 * len(dataset)) +val_size = len(dataset) - train_size + +train_subset, val_subset = random_split(dataset, [train_size, val_size]) +train_dataset = TransformedSubset(train_subset, transform_augemnt) +val_dataset = TransformedSubset(val_subset, transform) + + +train_loader = DataLoader(train_dataset, batch_size=256, shuffle=True) +val_loader = DataLoader(val_dataset, batch_size=256, shuffle=False) + +device = torch.device("cpu") if not torch.cuda.is_available() else torch.device("cuda:0") +print("Using device", device) + +model = ObjectDetectionCNN(c_in=3, c_hidden=32, c_out=2, layers=10) +model.to(device) +optimizer = torch.optim.Adam(model.parameters(), lr=1e-3, weight_decay=1e-4) +loss_module = nn.CrossEntropyLoss(weight=torch.tensor([1/3, 1], device=device, dtype=torch.float32)) + +#trainNormalDataset(model=model, loss_module=loss_module, train_loader=train_loader, val_loader=val_loader, +# optimizer=optimizer, SAVE_PATH=SAVE_PATH, saving=True, device=device, model_name="face_detection") + +#exit() + +image = read_image("data/faces/train/_url-http_3A_2F_2Fdingyue-ws-126-net_2F2023_2F0201_2F6a17ca80j00rpd7yx00bmd000dw00gop_jpg.rf.Setza3JikTSzG04c0Fd1.jpg") +image = image.float() / 255.0 +visualizeImage(image) + +eval(model=model, image=image, BUILD_PATH=os.path.join(SAVE_PATH, "face_detection", "face_detection"), device=device, minConf=0.97) \ No newline at end of file diff --git a/api/server.py b/api/src/server.py similarity index 97% rename from api/server.py rename to api/src/server.py index 2ea7237..1c5f3b4 100644 --- a/api/server.py +++ b/api/src/server.py @@ -12,11 +12,11 @@ from PIL import Image import io import torch.nn.functional as F -from bird_cnn import Bird_CNN +from bird_cnn.bird_cnn import Bird_CNN import threading -BUILD_PATH = "./build_models" +BUILD_PATH = "../build_models" IMAGE_SIZE = 64 sem_ai = threading.Semaphore(1) @@ -78,7 +78,7 @@ async def predict(file: UploadFile = File(...)): } -load_dotenv("database/.env") +load_dotenv("../database/.env") API_KEY = os.getenv("API_KEY") def get_api_key(authorization: str = Header(None)): diff --git a/api/util.py b/api/src/util.py similarity index 58% rename from api/util.py rename to api/src/util.py index 520f697..4ea7e6c 100644 --- a/api/util.py +++ b/api/src/util.py @@ -34,4 +34,28 @@ def visualizeData(dataset): plt.title(f"Label: {labels[i].item()}") plt.axis("off") - plt.show() \ No newline at end of file + plt.show() + +def visualizeImage(image): + image = image.permute(1, 2, 0) + + plt.figure(figsize=(3,3)) + plt.imshow(image) + plt.axis("off") + + plt.show() + +def iou(boxA, boxB): + xA = max(boxA[0], boxB[0]) + yA = max(boxA[1], boxB[1]) + xB = min(boxA[2], boxB[2]) + yB = min(boxA[3], boxB[3]) + + inter_area = max(0, xB - xA) * max(0, yB - yA) + + boxA_area = (boxA[2]-boxA[0]) * (boxA[3]-boxA[1]) + boxB_area = (boxB[2]-boxB[0]) * (boxB[3]-boxB[1]) + + union = boxA_area + boxB_area - inter_area + + return inter_area / union if union > 0 else 0 \ No newline at end of file diff --git a/api/src/yolo/train_yolo_faces.py b/api/src/yolo/train_yolo_faces.py new file mode 100644 index 0000000..5f49384 --- /dev/null +++ b/api/src/yolo/train_yolo_faces.py @@ -0,0 +1,55 @@ +import os + +from torchvision import transforms +from tqdm import tqdm +from yolo_model import train, Yolo_model +from yolo_loss import YoloLoss +from cocoDetectionDataset import CocoDetectionDatasetResized +from util import TransformedSubset, visualizeImage +from torch.utils.data import DataLoader, random_split +from torchvision import datasets +import torch +import torch.nn as nn +from torchvision.io import read_image +from torch.utils.data import WeightedRandomSampler + + + +SAVE_PATH = "./saved_models" +IMAGE_SIZE = 128 +GRID = 9 +BATCH_SIZE = 1 + +transform = transforms.Compose([ + transforms.ToTensor() +]) + +dataset = CocoDetectionDatasetResized( + image_dir="data/faces/train", + annotation_path="data/faces/train/_annotations.coco.json", + img_size=IMAGE_SIZE, + transforms=transform +) + +train_size = int(0.8 * len(dataset)) +val_size = len(dataset) - train_size + +train_subset, val_subset = random_split(dataset, [train_size, val_size]) + + +train_loader = DataLoader(train_subset, batch_size=BATCH_SIZE, shuffle=True, collate_fn=lambda x: tuple(zip(*x))) +val_loader = DataLoader(val_subset, batch_size=BATCH_SIZE, shuffle=False, collate_fn=lambda x: tuple(zip(*x))) + +device = torch.device("cpu") if not torch.cuda.is_available() else torch.device("cuda:0") +print("Using device", device) + +model = Yolo_model(c_in=3, c_hidden=32, boxes=2, img_size=IMAGE_SIZE, grid=GRID, labels=1) +model.to(device) +optimizer = torch.optim.Adam(model.parameters(), lr=1e-3, weight_decay=1e-4) +loss_module = YoloLoss() + +train(model=model, loss_module=loss_module, train_loader=train_loader, val_loader=val_loader, + optimizer=optimizer, SAVE_PATH=SAVE_PATH, saving=True, device=device, + model_name="face_detection_yolo", img_size=IMAGE_SIZE, num_classes=1, grid=GRID) + +exit() \ No newline at end of file diff --git a/api/src/yolo/yolo_loss.py b/api/src/yolo/yolo_loss.py new file mode 100644 index 0000000..0d3bd18 --- /dev/null +++ b/api/src/yolo/yolo_loss.py @@ -0,0 +1,107 @@ +import torch +import torch.nn as nn + +class YoloLoss(nn.Module): + def __init__(self): + super(YoloLoss, self).__init__() + + @staticmethod + def is_center_in_grid_cell(x, y, img_w, img_h, S, cell_i, cell_j): + cell_w = img_w / S + cell_h = img_h / S + + # find which cell the center belongs to + gt_cell_i = int(y / cell_h) + gt_cell_j = int(x / cell_w) + + return (gt_cell_i == cell_i) and (gt_cell_j == cell_j) + + @staticmethod + def save_sqrt(i): + return torch.sqrt(torch.clamp(i, min=1e-6)) + + @staticmethod + def iou(boxA, boxB): + xA = torch.max(boxA[0], boxB[0]) + yA = torch.max(boxA[1], boxB[1]) + xB = torch.min(boxA[2], boxB[2]) + yB = torch.min(boxA[3], boxB[3]) + + inter_area = torch.clamp(xB - xA, min=0) * torch.clamp(yB - yA, min=0) + + boxA_area = (boxA[2] - boxA[0]) * (boxA[3] - boxA[1]) + boxB_area = (boxB[2] - boxB[0]) * (boxB[3] - boxB[1]) + + union = boxA_area + boxB_area - inter_area + + return inter_area / (union + 1e-6) + + @staticmethod + def xywh_to_xyxy(box): + x, y, w, h = box + + return torch.stack([ + x - w / 2, + y - h / 2, + x + w / 2, + y + h / 2 + ]) + + def forward(self, pred, target, labels, cell_size, img_w, img_h, factor_no_object = 0.5, factor_object = 5): + #pred: [SxSx(B(x,y,w,h,con)C)] + #target: [B(x, y, w, h, C)] + + for s0 in range(pred.shape[0]): + for s1 in range(pred.shape[1]): + boxes = pred[s0, s1, :(-1*labels)] + pred_label = pred[s0, s1, -1:] + + ground_truth_box = None + for b in range(target.shape[0] // (4 + labels)): + box = target[(b)*(4 + labels):(b+1)*(4 + labels)] + if self.is_center_in_grid_cell(box[0]+box[2]/2, box[1]+box[3]/2, img_w, img_h, cell_size, s0, s1): + ground_truth_box = box[:4] + ground_truth_label = box[4:] + break + + + coord_loss = torch.tensor(0., device=pred.device) + size_loss = torch.tensor(0., device=pred.device) + prob_loss = torch.tensor(0., device=pred.device) + label_loss = torch.tensor(0., device=pred.device) + + if ground_truth_box is not None: + + iou_stats = [] + + for b in range(boxes.shape[0] // 5): + box = boxes[(b)*5:(b+1)*5] + + iou_stats.append(self.iou(self.xywh_to_xyxy(box[0:4]), self.xywh_to_xyxy(ground_truth_box[0:4]))) + iou_tensor = torch.stack(iou_stats) + responsible = torch.argmax(iou_tensor) + + for b in range(boxes.shape[0] // 5): + box = boxes[(b)*5:(b+1)*5] + + is_responsible = (b == responsible).float() + + coord_loss += factor_object * is_responsible * ((box[0] - ground_truth_box[0])**2 + (box[1] - ground_truth_box[1])**2) + size_loss += factor_object * is_responsible * ((self.save_sqrt(box[2]) - self.save_sqrt(ground_truth_box[2]))**2 + + (self.save_sqrt(box[3]) - self.save_sqrt(ground_truth_box[3]))**2) + + prob_loss += is_responsible * (box[4] - 1)**2 + prob_loss += factor_no_object * (1 - is_responsible) * box[4]**2 + + + for i in range(pred_label.shape[0]): + label_loss += ((pred_label[i] - ground_truth_label[i])**2) / pred_label.shape[0] + + + else: + for b in range(boxes.shape[0] // 5): + box = boxes[(b)*5:(b+1)*5] + + prob_loss += factor_no_object * box[4] ** 2 + + return (label_loss + prob_loss + coord_loss + size_loss) / pred.shape[0]**2 diff --git a/api/src/yolo/yolo_model.py b/api/src/yolo/yolo_model.py new file mode 100644 index 0000000..d95a8e7 --- /dev/null +++ b/api/src/yolo/yolo_model.py @@ -0,0 +1,134 @@ +import os +import cv2 +import torch +from tqdm import tqdm +import torch.nn as nn +import torch.nn.functional as F +import torchvision.transforms.functional as TF +import time + + +class Yolo_model(nn.Module): + def __init__(self, c_in, c_hidden, boxes, img_size, grid, labels): + super().__init__() + + self.grid = grid + + self.model = nn.Sequential( + nn.Conv2d(in_channels=c_in, out_channels=c_hidden, kernel_size=7, padding=3), + nn.LeakyReLU(), + nn.Conv2d(in_channels=c_hidden, out_channels=c_hidden, kernel_size=3, padding=1), + nn.LeakyReLU(), + nn.Conv2d(in_channels=c_hidden, out_channels=c_hidden, kernel_size=3, padding=1), + nn.LeakyReLU(), + nn.Conv2d(in_channels=c_hidden, out_channels=c_hidden, kernel_size=3, padding=1), + nn.LeakyReLU(), + nn.Conv2d(in_channels=c_hidden, out_channels=c_hidden, kernel_size=3, padding=1), + nn.LeakyReLU(), + nn.Conv2d(in_channels=c_hidden, out_channels=c_hidden, kernel_size=3, padding=1), + nn.LeakyReLU(), + nn.Flatten(), + nn.Linear(in_features=img_size*img_size*c_hidden, out_features=grid*grid*(boxes*5+labels)) + ) + + def forward(self, x): + batch = x.shape[0] + x = self.model(x) + x = x.reshape(batch, self.grid, self.grid, -1) + return x + +def create_stack(images, annotations, num_classes, device): + data = [] + labels = [] + for i in range(len(images)): + image = images[i] + + annotation = annotations[i] + + label_train = [] + for box, label in tuple(zip(annotation["boxes"], annotation["labels"])): + box = box + label = 0 + one_hot = F.one_hot(torch.tensor(label, dtype=torch.long), num_classes=num_classes) + label_train.append(torch.cat((box, one_hot), dim=0)) + + label_torch = torch.stack(label_train).reshape(-1).to(device) + labels.append(label_torch) + data.append(image) + + return data, labels + + +def train(model, num_classes, loss_module, train_loader, val_loader, optimizer, SAVE_PATH, model_name, grid, saving=True, device="cpu", img_size=64): + best_val = torch.finfo(torch.float32).max + + for epoch in range(200): + ############ + # Training # + ############ + model.train() + + count, lossCount = 0, 0. + for images, annotations in tqdm(train_loader, desc=f"Train", leave=False): + data, labels = create_stack(images, annotations, num_classes, device) + + data = torch.stack(data).to(device) + + prediction = model(data) + + loss = [] + for pred, label in tuple(zip(prediction, labels)): + loss.append(loss_module(pred = pred, target= label, labels=num_classes, cell_size=grid, img_w = img_size, img_h = img_size)) + loss = torch.stack(loss).to(device) + loss = loss.sum() + lossCount += loss.item() + + optimizer.zero_grad() + + loss.backward() + + optimizer.step() + + count += data.size(0) + + train_loss = lossCount / count + + torch.cuda.empty_cache() + + ############## + # Validation # + ############## + model.eval() + + count, lossCount = 0, 0. + for images, annotations in tqdm(val_loader, desc=f"Test", leave=False): + with torch.no_grad(): + data, labels = create_stack(images, annotations, num_classes, device) + + data = torch.stack(data).to(device) + + prediction = model(data) + + loss = [] + for pred, label in tuple(zip(prediction, labels)): + loss.append(loss_module(pred = pred, target= label, labels=num_classes, cell_size=grid, img_w = img_size, img_h = img_size)) + loss = torch.stack(loss).to(device) + + + lossCount += loss.sum().item() + + count += data.size(0) + + val_loss = lossCount / count + + if(saving and best_val > val_loss): + best_val = val_loss + save_dir = os.path.join(SAVE_PATH, model_name) + os.makedirs(save_dir, exist_ok=True) + + save_path = os.path.join(save_dir, model_name) + torch.save(model.state_dict(), save_path) + + print(f"epoch: {epoch+1} | train loss: {int(train_loss * 1000) / 100} | val loss: {int(val_loss * 1000) / 100}") + torch.cuda.empty_cache() + return best_val \ No newline at end of file diff --git a/api/start_server.bat b/api/start_server.bat index dc9811c..a021550 100644 --- a/api/start_server.bat +++ b/api/start_server.bat @@ -1 +1 @@ -python -m uvicorn server:app --reload \ No newline at end of file +python -m uvicorn src/server:app --reload \ No newline at end of file diff --git a/api/train_cifar.py b/api/train_cifar.py deleted file mode 100644 index 0ee2acb..0000000 --- a/api/train_cifar.py +++ /dev/null @@ -1,41 +0,0 @@ -import torch -import torch.nn as nn -from torchvision import datasets, transforms -from torch.utils.data import DataLoader, random_split - -from bird_cnn import Bird_CNN, sample, trainCNN - -SAVE_PATH = "./saved_models" - -transform = transforms.Compose([ - transforms.ToTensor(), - transforms.Normalize((0.5, 0.5, 0.5), - (0.5, 0.5, 0.5)) -]) - -train_dataset = datasets.CIFAR10( - root="./data/cifar10", - train=True, - download=True, - transform=transform -) - -val_dataset = datasets.CIFAR10( - root="./data/cifar10", - train=False, - download=True, - transform=transform -) - -train_loader = DataLoader(train_dataset, batch_size=4, shuffle=True) -val_loader = DataLoader(val_dataset, batch_size=4, shuffle=False) - -device = torch.device("cpu") if not torch.cuda.is_available() else torch.device("cuda:0") -print("Using device", device) - -model = Bird_CNN(c_in=3, c_hidden=16, c_out=10) -model.to(device) -optimizer = torch.optim.Adam(model.parameters(), lr=1e-3, weight_decay=1e-4) -loss_module = nn.CrossEntropyLoss() - -trainCNN(model, optimizer, loss_module, train_loader, val_loader, device, 50, SAVE_PATH=SAVE_PATH, save=False) \ No newline at end of file diff --git a/api/train_mnist.py b/api/train_mnist.py deleted file mode 100644 index 5e8a82d..0000000 --- a/api/train_mnist.py +++ /dev/null @@ -1,41 +0,0 @@ -import torch -import torch.nn as nn -from torchvision import datasets, transforms -from torch.utils.data import DataLoader, random_split - -from bird_cnn import Bird_CNN, sample, trainCNN - -SAVE_PATH = "./saved_models" - -transform = transforms.Compose([ - transforms.ToTensor(), - transforms.Normalize((0.5), - (0.5)) -]) - -train_dataset = datasets.MNIST( - root="./data/mnist", - train=True, - download=True, - transform=transform -) - -val_dataset = datasets.MNIST( - root="./data/mnist", - train=False, - download=True, - transform=transform -) - -train_loader = DataLoader(train_dataset, batch_size=32, shuffle=True) -val_loader = DataLoader(val_dataset, batch_size=32, shuffle=False) - -device = torch.device("cpu") if not torch.cuda.is_available() else torch.device("cuda:0") -print("Using device", device) - -model = Bird_CNN(c_in=1, c_hidden=4, c_out=10) -model.to(device) -optimizer = torch.optim.Adam(model.parameters(), lr=1e-3, weight_decay=1e-4) -loss_module = nn.CrossEntropyLoss() - -trainCNN(model, optimizer, loss_module, train_loader, val_loader, device, 50, SAVE_PATH=SAVE_PATH, save=False) \ No newline at end of file