{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<center>\n    <h1>Diagnosis of breast cancer-Model Development </h1>\n<center>","metadata":{}},{"cell_type":"markdown","source":"<center>\n<img src=\"https://www.breastcancer.org/_next/image?url=https%3A%2F%2Fimages.ctfassets.net%2Fzzorm7zihro2%2Fbfe84cf4-d721-4a39-9da9-f72b8c58eda8%2F242be4758e0cbc27f48a03aa706fd4a4%2FMammographyComparison-2048x1463_labels.png&w=1920&q=75\">\n</center>\n","metadata":{}},{"cell_type":"markdown","source":"# Description\nIn this notebook, me and my team build a **CustomDataset**, a **Dataloader** and a **Deep learning** model. This is part of a problem-solving process that uses deep learning to diagnose breast cancer.\n\n### **Data**\n\n- #### To solve this problem, in addition to RSNA's image data, we also use 3 more datasets: INbreast, MIAS, and DDSM.\n\n### **Model** \n\n- #### In this notebook, we use a basic CNN model combined with fully connected. Detailed model pictures are below.\n\n<center>\n<img src=\"https://scontent.fdad3-4.fna.fbcdn.net/v/t1.15752-9/317923657_2181252688743181_864461044515013559_n.png?_nc_cat=101&ccb=1-7&_nc_sid=ae9488&_nc_ohc=gT6Cj4Ai4X4AX-_57wx&_nc_ht=scontent.fdad3-4.fna&oh=03_AdSnbNZxk7QiPP8rWt5GuSAbobySytq-uOeHzR4aLCnGEQ&oe=63F24051\">\n</center>\n\n\n### Reference\n\nThe architecture mentioned above is the one we refer to [here](https://s3.us-west-2.amazonaws.com/secure.notion-static.com/502af837-13ee-4354-b419-982540af6598/1-s2.0-S0378475422002506-main_%281%29.pdf?X-Amz-Algorithm=AWS4-HMAC-SHA256&X-Amz-Content-Sha256=UNSIGNED-PAYLOAD&X-Amz-Credential=AKIAT73L2G45EIPT3X45%2F20230120%2Fus-west-2%2Fs3%2Faws4_request&X-Amz-Date=20230120T160710Z&X-Amz-Expires=86400&X-Amz-Signature=df495c3bce0ed9e515bf12a1dae101aa77eb70feacf6127b38c7a9967bbab052&X-Amz-SignedHeaders=host&response-content-disposition=filename%3D%221-s2.0-S0378475422002506-main%2520%281%29.pdf%22&x-id=GetObject).\n\n\n\n\n","metadata":{}},{"cell_type":"markdown","source":"# 1. Install lib","metadata":{}},{"cell_type":"code","source":"!pip install /kaggle/input/rsnawhl/{pydicom-2.3.0-py3-none-any.whl,pylibjpeg-1.4.0-py3-none-any.whl,python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl}","metadata":{"execution":{"iopub.status.busy":"2023-02-06T13:54:53.747214Z","iopub.execute_input":"2023-02-06T13:54:53.747581Z","iopub.status.idle":"2023-02-06T13:55:26.7425Z","shell.execute_reply.started":"2023-02-06T13:54:53.747549Z","shell.execute_reply":"2023-02-06T13:55:26.741313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install /kaggle/input/nvidia-dali-wheel/nvidia_dali_nightly_cuda110-1.22.0.dev20221213-6757685-py3-none-manylinux2014_x86_64.whl\n!pip install /kaggle/input/nvidia-dali-wheel/dicomsdl-0.109.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl","metadata":{"execution":{"iopub.status.busy":"2023-02-06T13:55:26.745133Z","iopub.execute_input":"2023-02-06T13:55:26.745619Z","iopub.status.idle":"2023-02-06T13:56:39.88656Z","shell.execute_reply.started":"2023-02-06T13:55:26.745555Z","shell.execute_reply":"2023-02-06T13:56:39.885371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pip install torchsummary wandb pydicom","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-02-06T13:56:39.888825Z","iopub.execute_input":"2023-02-06T13:56:39.889976Z","iopub.status.idle":"2023-02-06T13:56:39.895414Z","shell.execute_reply.started":"2023-02-06T13:56:39.889932Z","shell.execute_reply":"2023-02-06T13:56:39.894387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pip install dicomsdl","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-02-06T13:56:39.898427Z","iopub.execute_input":"2023-02-06T13:56:39.898866Z","iopub.status.idle":"2023-02-06T13:56:39.909586Z","shell.execute_reply.started":"2023-02-06T13:56:39.898832Z","shell.execute_reply":"2023-02-06T13:56:39.908633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Import lib","metadata":{}},{"cell_type":"code","source":"import cv2 \nimport torch\nimport imghdr\nimport pydicom\nimport plistlib\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nimport dicomsdl\nimport torch.nn as nn\nfrom PIL import Image\nimport os, sys, tarfile\nfrom skimage import filters\nfrom skimage.io import imread\nimport pytorch_lightning as pl\nfrom tqdm.notebook import tqdm\nimport torch.nn.functional as F\n# from torchsummary import summary\nfrom skimage.draw import polygon\nimport torchvision.transforms as T\nfrom torchvision.io import read_image\nfrom skimage.exposure import equalize_adapthist\nfrom torch.utils.data import DataLoader, Dataset, ConcatDataset\n\n# Data Augmentation for Image Preprocessing\nfrom albumentations.pytorch import ToTensorV2\nfrom albumentations import (ToFloat, Normalize, VerticalFlip, HorizontalFlip, Compose, Resize,\n                            RandomBrightnessContrast, HueSaturationValue, Blur, GaussNoise,\n                            Rotate, RandomResizedCrop, Cutout, ShiftScaleRotate, ToGray)\n\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2023-02-06T13:56:39.912874Z","iopub.execute_input":"2023-02-06T13:56:39.913142Z","iopub.status.idle":"2023-02-06T13:56:45.929263Z","shell.execute_reply.started":"2023-02-06T13:56:39.913119Z","shell.execute_reply":"2023-02-06T13:56:45.928282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Dataset and Dataloader","metadata":{}},{"cell_type":"markdown","source":"### INBREAST DATASET","metadata":{}},{"cell_type":"code","source":"class INBreastDataset(Dataset):\n\n    '''\n    INBreast dataset\n    '''\n    def __init__(self, csv_file: str, \n                 root_dir: str, \n                 transform=None):\n        '''\n            Args:\n            csv_file (string): Path to the csv file with annotations.\n            root_dir (string): Directory with all the images.\n            transform (callable, optional): Optional transform to be applied\n                on a sample.\n        '''\n        self.img_dir = csv_file\n        self.img_data = pd.read_csv(csv_file)\n        self.root = root_dir\n        self.transform = transform\n        \n    def __len__(self):\n        return len(self.img_data['Lesion Annotation Status'])\n    \n    def pre_img(self, img):\n        img_clahe = equalize_adapthist(img, clip_limit=0.04, nbins=256)\n        return img_clahe\n\n    def __getitem__(self, idx):\n        img_path = os.path.join(self.root, self.img_data.iloc[idx, -1].replace('dcm', 'png'))\n        img = cv2.imread(img_path, 0)\n        img = self.pre_img(img)\n        if self.transform != None:\n            image_trans = self.transform(image=img)['image']\n        else:\n            image_trans = img\n        #return\n        final_image = image_trans\n        label = self.img_data['Lesion Annotation Status'][idx]\n        return final_image, label","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-02-06T13:56:45.930734Z","iopub.execute_input":"2023-02-06T13:56:45.931576Z","iopub.status.idle":"2023-02-06T13:56:45.942881Z","shell.execute_reply.started":"2023-02-06T13:56:45.931536Z","shell.execute_reply":"2023-02-06T13:56:45.94135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### RSNA DATASET","metadata":{}},{"cell_type":"code","source":"class RSNADataset(Dataset):\n    \n    '''\n    RSNA Dataset\n    '''\n    def __init__(self,csv_file: str,\n                 root_dir : str,\n                 transform=None,is_train = True):\n        '''\n            Args:\n            csv_file (string): Path to the csv file with annotations.\n            root_dir (string): Directory with all the images.\n            transform (callable, optional): Optional transform to be applied\n                on a sample.\n        '''\n        self.dataframe = pd.read_csv(csv_file)\n        self.root_dir = root_dir\n        self.is_train = is_train\n        \n        if is_train:\n            self.transform = transform\n        else:\n            self.transform = Compose([Resize(height=227,width=227,always_apply=True),\n                                      ToTensorV2()])\n    \n    def __len__(self):\n        return len(self.dataframe)\n    \n    \n    def get_cdf(self, img):\n        m = int(np.max(img))\n        hist = np.histogram(img, bins=m+1, range=(0, m+1))[0]\n        hist = hist/img.size\n        cdf = np.cumsum(hist)\n        return cdf\n\n    # numpy\n    def histogram_equalization(self, img):\n        m = int(np.max(img))\n        hist = np.histogram(img, bins=m+1, range=(0, m+1))[0]\n        # bước 1: tính pdf\n        hist = hist/img.size\n        # bước 2: tính cdf\n        cdf = np.cumsum(hist)\n        # bước 3: lập bảng thay thế\n        s_k = (255 * cdf)\n        # ảnh mới\n        img_new = np.array([s_k[i] for i in img.ravel()]).reshape(img.shape)\n        return img_new\n\n\n    def HE(self, img):\n        '''\n            input: gray image\n            dtype: np.array\n        '''\n        img_equ= self.histogram_equalization(img)\n        return img_equ\n\n    def pre_img(self, img):\n#         clahe = cv2.createCLAHE(clipLimit = 20)\n#         img_clahe = clahe.apply(img) + 30\n        img_clahe = equalize_adapthist(img, clip_limit=0.04, nbins=256)\n#         HE_img = self.HE(img_clahe)\n#         print(HE_img)\n        return img_clahe\n\n    def __getitem__(self,index):\n        if self.is_train:\n            if self.dataframe.iloc[index]['cancer']==0:\n                image_path = self.root_dir + '/nocancer/'+str(self.dataframe.iloc[index].patient_id) + \"_\" + str(self.dataframe.iloc[index].image_id) + \".png\"\n            else:\n                image_path = self.root_dir + '/cancer/'+str(self.dataframe.iloc[index].patient_id) + \"_\" + str(self.dataframe.iloc[index].image_id) + \".png\"\n        else:\n            image_path = self.root_dir +'/'+ str(self.dataframe.iloc[index].patient_id) + \"/\" + str(self.dataframe.iloc[index].image_id) + \".png\"\n        image = cv2.imread(image_path,0)\n        image = self.pre_img(image)\n        \n        #transform image\n        # return a dict have key 'image'\n        if self.transform != None:\n            image_trans = self.transform(image=image)['image']\n        else:\n            image_trans = image\n            \n        #return\n        final_image = image_trans\n        if self.is_train:\n            label = self.dataframe.iloc[index]['cancer'] \n            return final_image,label\n        else: \n            return final_image","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-02-06T13:56:45.944296Z","iopub.execute_input":"2023-02-06T13:56:45.947551Z","iopub.status.idle":"2023-02-06T13:56:45.963463Z","shell.execute_reply.started":"2023-02-06T13:56:45.947515Z","shell.execute_reply":"2023-02-06T13:56:45.962641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### MIAS DATASET","metadata":{}},{"cell_type":"code","source":"class MIASDataset(Dataset):\n    def __init__(self, \n                 csv_file : str,\n                 root: str,\n                 transform=None):\n        self.img_dir = root\n        self.total_imgs = pd.read_csv(csv_file)\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.total_imgs)\n\n    def pre_img(self, img):\n        img_clahe = equalize_adapthist(img, clip_limit=0.04, nbins=256)\n        return img_clahe\n    \n    def __getitem__(self, idx):\n        img_path = self.total_imgs['Images path'][idx]\n        image = cv2.imread(img_path, 0)\n        label = self.total_imgs['Labels'][idx]\n        image = self.pre_img(image)\n        if self.transform != None:\n            image_trans = self.transform(image=image)['image']\n        else:\n            image_trans = image\n        #return\n        final_image = image_trans\n#         print(image_trans.shape)\n        return final_image, label","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-02-06T13:56:45.96471Z","iopub.execute_input":"2023-02-06T13:56:45.965134Z","iopub.status.idle":"2023-02-06T13:56:45.979513Z","shell.execute_reply.started":"2023-02-06T13:56:45.965094Z","shell.execute_reply":"2023-02-06T13:56:45.978652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class clr:\n    START = '\\033[1m'\n    PURPLE = '\\033[95m'\n    CYAN = '\\033[96m'\n    DARKCYAN = '\\033[36m'\n    BLUE = '\\033[94m'\n    GREEN = '\\033[92m'\n    YELLOW = '\\033[93m'\n    RED = '\\033[91m'\n    BOLD = '\\033[1m'\n    UNDERLINE = '\\033[4m'\n    END = '\\033[0m'\n","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-02-06T13:56:45.980945Z","iopub.execute_input":"2023-02-06T13:56:45.981362Z","iopub.status.idle":"2023-02-06T13:56:45.991817Z","shell.execute_reply.started":"2023-02-06T13:56:45.981322Z","shell.execute_reply":"2023-02-06T13:56:45.990591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Building Datamodule ","metadata":{}},{"cell_type":"code","source":"class Datamodule(pl.LightningDataModule):\n    \n    '''\n        Datamodule to load data\n    '''\n    def __init__(self, \n              batch_size: int,\n              train_set,\n              test_set=None,\n              validate_set=None):\n        super().__init__()\n\n        self.train_set = train_set\n        self.test_set = test_set\n        self.validate_set = validate_set\n        self.batch_size = batch_size\n\n\n    def train_dataloader(self):\n        '''\n        DataLoader of train func\n        '''\n        return DataLoader(\n            self.train_set,\n            batch_size=self.batch_size,\n            shuffle=True, # shuffle dữ liệu để tạo sự đa dạng,\n            drop_last=True\n        )\n\n    def val_dataloader(self):\n        '''\n        DataLoader of validate func\n        '''\n        return DataLoader(\n            self.validate_set,\n            batch_size=self.batch_size,\n        )\n\n    def test_dataloader(self):\n        '''\n        DataLoader of test func\n        '''\n        return DataLoader(\n            self.test_set,\n            batch_size=self.batch_size,\n        )\n","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-02-06T13:56:45.998162Z","iopub.execute_input":"2023-02-06T13:56:45.999036Z","iopub.status.idle":"2023-02-06T13:56:46.010839Z","shell.execute_reply.started":"2023-02-06T13:56:45.999002Z","shell.execute_reply":"2023-02-06T13:56:46.009339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Model","metadata":{}},{"cell_type":"markdown","source":"### a. Model base","metadata":{}},{"cell_type":"code","source":"class CNN(pl.LightningModule):\n    def __init__(self,\n                width:int=227,\n                height:int=227):\n        super().__init__()\n        def layer(input_channel: int, \n                  output_channel: int, \n                  kernel_size_conv: int=3, \n                  padding: str='same', \n                  stride_conv: int=1, \n                  kernel_size_maxpool: int=2, \n                  stride_maxpool: int=2,\n                  normalize=True,\n                  maxpool=True):\n            layers = [nn.Conv2d(input_channel, \n                                output_channel, \n                                kernel_size_conv,\n                                stride_conv,\n                                padding\n                                )]\n            if normalize:\n                layers.append(nn.BatchNorm2d(output_channel))\n            layers.append(nn.ReLU())\n            if maxpool:\n                layers.append(nn.MaxPool2d(kernel_size=kernel_size_maxpool, \n                                           stride=stride_maxpool))\n            return layers\n        \n        self.model = nn.Sequential(\n                        *layer(input_channel=1,\n                               output_channel=16),\n                        *layer(input_channel=16,\n                               output_channel=32),\n                        *layer(input_channel=32,\n                               output_channel=64,\n                               maxpool=False)\n                    )\n        \n        self.relu = nn.ReLU()\n        self.fc1 = nn.Linear((width//4) * (height//4) * 64, 512)\n        self.fc2 = nn.Linear(512, 128)\n        self.fc3 = nn.Linear(128, 2)\n        self.softmax = nn.Softmax(dim=1)\n    def forward(self, x):\n#         print(type(x))\n        x = self.model(x.float())\n        x = torch.flatten(x, 1)\n        x = self.relu(self.fc1(x))\n        x = self.relu(self.fc2(x))\n#         print(x)\n        x = self.fc3(x)   \n        x = self.softmax(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2023-02-06T13:56:46.012443Z","iopub.execute_input":"2023-02-06T13:56:46.013165Z","iopub.status.idle":"2023-02-06T13:56:46.025068Z","shell.execute_reply.started":"2023-02-06T13:56:46.013132Z","shell.execute_reply":"2023-02-06T13:56:46.024242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### b. Model using pytorchlighning","metadata":{}},{"cell_type":"code","source":"class FullModel(pl.LightningModule):\n    def __init__(self, \n                 lr: float, \n                 total_steps: int,\n                 width:int=227,\n                 height:int=227\n                 ):\n        super().__init__()\n        self.lr = lr\n        self.model = CNN(width=width, height=height)\n        self.criterion = nn.CrossEntropyLoss()\n        self.total_steps = total_steps\n        self.save_hyperparameters()\n        self.dem = 0\n        \n    def configure_optimizers(self):\n#         optimizer = torch.optim.AdamW(self.parameters(), lr=self.lr, betas=(0.9, 0.98), weight_decay=1e-4)\n        optimizer = torch.optim.SGD(model.parameters(), lr=0.1)\n#         optimizer = torch.optim.RMSprop(self.parameters(), lr=0.0001)\n        scheduler = torch.optim.lr_scheduler.OneCycleLR(\n            optimizer, max_lr=self.lr, total_steps=self.total_steps, verbose=False,\n        )\n        scheduler = {\n            \"scheduler\": scheduler,\n            \"interval\": \"step\",  # or 'epoch'\n            \"frequency\": 1,\n        }\n        return [optimizer], [scheduler]\n    \n    def forward(self, x):\n        return self.model(x)\n\n    def training_step(self, batch, batch_idx):\n        '''\n        Hàm này nhận vào một batch\n        lấy kết quả của mô hình và tính loss\n        mô hình sẽ tự tính back propagation\n        -----------------------------------\n        This function takes a batch, gets the \n        result of the model, and calculates the loss. \n        The model will automatically calculate back propagation.\n        '''\n        img, labels = batch\n#         print(img[0].shape, self.dem)\n        outputs = self.model(img)\n        loss = self.criterion(outputs, labels)\n        self.log(\"train_loss\", loss.item())\n        #Accuracy\n        output = torch.argmax(outputs, dim=1)\n        correct = (output == labels).float().sum()\n#         print(\"train\", labels, output)\n        self.log(\"train_loss\", loss.item())\n        self.log(\"train_acc\", correct/len(labels))\n#         self.dem+=1\n        return loss\n\n    def validation_step(self, batch, batch_idx):\n        '''\n        Hàm này nhận vào một batch\n        lấy kết quả của mô hình và tính loss.\n        ---------------------------------------------\n        This function takes a batch, gets the \n        result of the model, and calculates the loss. \n        '''\n        img, labels = batch\n        outputs = self.model(img)\n        loss = self.criterion(outputs, labels)\n        #Accuracy\n        output = torch.argmax(outputs, dim=1)\n#         print(\"val\", labels, output)\n        correct = (output == labels).float().sum()\n\n        # lưu lại loss của validate\n        self.log(\"val_loss\", loss.item())\n        self.log(\"val_acc\", correct/len(labels))\n\n        return loss\n\n    def test_step(self, batch, batch_idx):\n        img, labels = batch\n        outputs = self.model(img)\n        loss = self.criterion(outputs, labels)\n\n        self.log(\"train_loss\", loss.item())\n\n        # lưu lại loss của test\n        self.log(\"test_loss\", loss.item())\n        self.log(\"test_acc\", acc.item())\n\n        return loss\n","metadata":{"execution":{"iopub.status.busy":"2023-02-06T13:56:46.026494Z","iopub.execute_input":"2023-02-06T13:56:46.027115Z","iopub.status.idle":"2023-02-06T13:56:46.047875Z","shell.execute_reply.started":"2023-02-06T13:56:46.027082Z","shell.execute_reply":"2023-02-06T13:56:46.046955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. LOGIN WANDB","metadata":{}},{"cell_type":"markdown","source":"#### A tool for visualizing and tracking your machine learning experiments. This repo contains the CLI and Python API.","metadata":{}},{"cell_type":"code","source":"# pip install wandb --upgrade","metadata":{"execution":{"iopub.status.busy":"2023-02-06T13:56:46.049306Z","iopub.execute_input":"2023-02-06T13:56:46.050005Z","iopub.status.idle":"2023-02-06T13:56:46.06273Z","shell.execute_reply.started":"2023-02-06T13:56:46.049972Z","shell.execute_reply":"2023-02-06T13:56:46.061679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Log in to Wandb if you have an account","metadata":{}},{"cell_type":"code","source":"# from kaggle_secrets import UserSecretsClient\n# user_secrets = UserSecretsClient()\n# secret_value_0 = user_secrets.get_secret(\"wandb\")\n# !wandb login $secret_value_0","metadata":{"execution":{"iopub.status.busy":"2023-02-06T13:56:46.064197Z","iopub.execute_input":"2023-02-06T13:56:46.064931Z","iopub.status.idle":"2023-02-06T13:56:46.073654Z","shell.execute_reply.started":"2023-02-06T13:56:46.064899Z","shell.execute_reply":"2023-02-06T13:56:46.072922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pd.read_csv(\"/kaggle/input/breastcancer/RSNA_metadata.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-02-06T13:56:46.075283Z","iopub.execute_input":"2023-02-06T13:56:46.075964Z","iopub.status.idle":"2023-02-06T13:56:46.08434Z","shell.execute_reply.started":"2023-02-06T13:56:46.075921Z","shell.execute_reply":"2023-02-06T13:56:46.083392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### a. Reading dataset","metadata":{}},{"cell_type":"code","source":"# root_inbreast = \"/kaggle/input/inbreast-crop-png/ALL-IMGS\"\n# csv_inbreast = \"/kaggle/input/breastcancer/INBreast_config.csv\"\n# csv_RSNA = '/kaggle/input/breastcancer/RSNA_metadata.csv'\n# root_RSNA = '/kaggle/input/rsna-256px-croped/RSNA_Crop'\n\n# root_test_RSNA = '/kaggle/input/test-rsna-dataset'\ncsv_test_RSNA = '/kaggle/input/rsna-breast-cancer-detection/test.csv'\n# root_MIAS='/kaggle/input/mias-png'\n# csv_MIAS= '/kaggle/input/breastcancer/mias_metadata.csv'\n\n# transform = Compose([Resize(height=227,width=227,always_apply=True),\n# #                     Normalize(mean=0.449,std=0.226),\n#                     HorizontalFlip(),\n#                     VerticalFlip(),\n#                     Rotate(),\n#                     ToTensorV2()])\n\n# train_inbreast = INBreastDataset(csv_inbreast, root_inbreast, transform)\n# train_rsna = RSNADataset(csv_RSNA,root_RSNA,transform)\n# train_mias = MIASDataset(csv_MIAS, root_MIAS, transform)\n# test_rsna = RSNADataset(csv_test_RSNA, root_test_RSNA, transform=transform, is_train=False)\n# print(len(train_inbreast), len(train_rsna), len(test_rsna))","metadata":{"execution":{"iopub.status.busy":"2023-02-06T13:56:46.085851Z","iopub.execute_input":"2023-02-06T13:56:46.086515Z","iopub.status.idle":"2023-02-06T13:56:46.096443Z","shell.execute_reply.started":"2023-02-06T13:56:46.08648Z","shell.execute_reply":"2023-02-06T13:56:46.095457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# mias = pd.read_csv(csv_MIAS)\n# inbrest = pd.read_csv(csv_inbreast)\n# rsna = pd.read_csv(csv_RSNA)\n# np.unique(inbrest['Lesion Annotation Status'], return_counts=True)\n","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-02-06T13:56:46.098021Z","iopub.execute_input":"2023-02-06T13:56:46.098721Z","iopub.status.idle":"2023-02-06T13:56:46.111519Z","shell.execute_reply.started":"2023-02-06T13:56:46.098688Z","shell.execute_reply":"2023-02-06T13:56:46.110384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# np.unique(rsna.cancer, return_counts=True)","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-02-06T13:56:46.113052Z","iopub.execute_input":"2023-02-06T13:56:46.11366Z","iopub.status.idle":"2023-02-06T13:56:46.121902Z","shell.execute_reply.started":"2023-02-06T13:56:46.113625Z","shell.execute_reply":"2023-02-06T13:56:46.120894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# np.unique(mias.Labels, return_counts=True)","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-02-06T13:56:46.123477Z","iopub.execute_input":"2023-02-06T13:56:46.124101Z","iopub.status.idle":"2023-02-06T13:56:46.13181Z","shell.execute_reply.started":"2023-02-06T13:56:46.124068Z","shell.execute_reply":"2023-02-06T13:56:46.130855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_dataset = ConcatDataset([train_inbreast, train_mias])\n# test_dataset = ConcatDataset([test_rsna])\n# dm = Datamodule(32, train_dataset, train_rsna, train_rsna)","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-02-06T13:56:46.133387Z","iopub.execute_input":"2023-02-06T13:56:46.134069Z","iopub.status.idle":"2023-02-06T13:56:46.142644Z","shell.execute_reply.started":"2023-02-06T13:56:46.134035Z","shell.execute_reply":"2023-02-06T13:56:46.141541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### b. Visualize the image after preprocessing.","metadata":{}},{"cell_type":"code","source":"def visulize(dataset, mode='train'):\n    plt.figure(figsize=(15, 15))\n    if mode == 'train':\n        for i, (img, label) in enumerate(dataset):\n        #     print(img.permute(1, 2, 0).shape, label)\n        #     print(img.shape)\n            plt.subplot(1,10,i+1)\n            plt.imshow(img.squeeze())\n            plt.axis('off')\n            plt.subplots_adjust(wspace=None, hspace=None)\n            plt.title(label)\n            if i == 9:\n                break\n    else:\n        for i, (img) in enumerate(dataset):\n        #     print(img.permute(1, 2, 0).shape, label)\n        #     print(img.shape)\n            plt.subplot(1,4,i+1)\n            plt.imshow(img.squeeze())\n            plt.axis('off')\n            plt.subplots_adjust(wspace=None, hspace=None)\n            if i == 3:\n                break\n# visulize(train_dataset, mode='train')","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-02-06T13:56:46.144434Z","iopub.execute_input":"2023-02-06T13:56:46.145091Z","iopub.status.idle":"2023-02-06T13:56:46.153857Z","shell.execute_reply.started":"2023-02-06T13:56:46.145059Z","shell.execute_reply":"2023-02-06T13:56:46.152891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# visulize(test_dataset, mode='test')","metadata":{"execution":{"iopub.status.busy":"2023-02-06T13:56:46.155423Z","iopub.execute_input":"2023-02-06T13:56:46.156106Z","iopub.status.idle":"2023-02-06T13:56:46.169414Z","shell.execute_reply.started":"2023-02-06T13:56:46.156072Z","shell.execute_reply":"2023-02-06T13:56:46.168478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### c. Create model and training","metadata":{}},{"cell_type":"code","source":"max_epochs = 100\nlr = 0.05\n# total_step = len(dm.train_dataloader()) * max_epochs\n# model = FullModel(lr, \n#                   0,\n#                   width=227,\n#                   height=227)","metadata":{"execution":{"iopub.status.busy":"2023-02-06T13:56:46.170948Z","iopub.execute_input":"2023-02-06T13:56:46.171556Z","iopub.status.idle":"2023-02-06T13:56:46.180621Z","shell.execute_reply.started":"2023-02-06T13:56:46.171523Z","shell.execute_reply":"2023-02-06T13:56:46.179655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# wandb_logger = pl.loggers.WandbLogger(\n#     project=\"BREAST-CANCER\", name='model_CNN_ver3', log_model=\"all\"\n# )\n# lr_monitor = pl.callbacks.LearningRateMonitor(logging_interval=\"step\")\n\n# trainer = pl.Trainer(\n#     logger=wandb_logger,\n#     callbacks=[lr_monitor],\n#     max_epochs=max_epochs,\n#     accelerator=\"auto\",\n#     log_every_n_steps=1\n#     gradient_clip_val=4\n# )\n\n# trainer.fit(model, datamodule=dm)#, ckpt_path='/kaggle/working/depth_estimation/26ozxilx/checkpoints/epoch=1-step=2850.ckpt')","metadata":{"execution":{"iopub.status.busy":"2023-02-06T13:56:46.182305Z","iopub.execute_input":"2023-02-06T13:56:46.183073Z","iopub.status.idle":"2023-02-06T13:56:46.19201Z","shell.execute_reply.started":"2023-02-06T13:56:46.183041Z","shell.execute_reply":"2023-02-06T13:56:46.191102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### d. Save checkpoint","metadata":{}},{"cell_type":"code","source":"# trainer.save_checkpoint(\"best_model.ckpt\")","metadata":{"execution":{"iopub.status.busy":"2023-02-06T13:56:46.193594Z","iopub.execute_input":"2023-02-06T13:56:46.194291Z","iopub.status.idle":"2023-02-06T13:56:46.202909Z","shell.execute_reply.started":"2023-02-06T13:56:46.194258Z","shell.execute_reply":"2023-02-06T13:56:46.201673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### e. Predict","metadata":{}},{"cell_type":"code","source":"PATH = '/kaggle/input/ckpt-rsna/model_ver3.1.ckpt'\n\nmodel = FullModel(lr, \n                  0,\n                  width=227,\n                  height=227).load_from_checkpoint(PATH)","metadata":{"execution":{"iopub.status.busy":"2023-02-06T13:56:46.204404Z","iopub.execute_input":"2023-02-06T13:56:46.205008Z","iopub.status.idle":"2023-02-06T13:56:59.361993Z","shell.execute_reply.started":"2023-02-06T13:56:46.204975Z","shell.execute_reply":"2023-02-06T13:56:59.360915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model.eval()","metadata":{"execution":{"iopub.status.busy":"2023-02-06T13:56:59.363541Z","iopub.execute_input":"2023-02-06T13:56:59.364226Z","iopub.status.idle":"2023-02-06T13:56:59.369063Z","shell.execute_reply.started":"2023-02-06T13:56:59.364187Z","shell.execute_reply":"2023-02-06T13:56:59.368108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# trainer.validate(model, dataloaders=dm)","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-02-06T13:56:59.37604Z","iopub.execute_input":"2023-02-06T13:56:59.376804Z","iopub.status.idle":"2023-02-06T13:56:59.383541Z","shell.execute_reply.started":"2023-02-06T13:56:59.376765Z","shell.execute_reply":"2023-02-06T13:56:59.382423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def pre_img(img):\n#     img_clahe = equalize_adapthist(img, clip_limit=0.04, nbins=256)\n#     return img_clahe\n# # img = cv2.imread(\"/kaggle/input/rsna-256px-croped/RSNA_Crop/cancer/10130_1360338805.png\", 0)\n# img = cv2.imread(\"/kaggle/input/test-rsna-dataset/10008/1591370361.png\", 0)\n# img = pre_img(img)\n# transform = Compose([Resize(height=227,width=227,always_apply=True),\n#                                       ToTensorV2()])\n# img = transform(image=img)['image']\n# # plt.imshow(img.squeeze())","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-02-06T13:56:59.386998Z","iopub.execute_input":"2023-02-06T13:56:59.387317Z","iopub.status.idle":"2023-02-06T13:56:59.396497Z","shell.execute_reply.started":"2023-02-06T13:56:59.387283Z","shell.execute_reply":"2023-02-06T13:56:59.395468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# visulize(test_dataset, mode='test')","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-02-06T13:56:59.398025Z","iopub.execute_input":"2023-02-06T13:56:59.39843Z","iopub.status.idle":"2023-02-06T13:56:59.411643Z","shell.execute_reply.started":"2023-02-06T13:56:59.398393Z","shell.execute_reply":"2023-02-06T13:56:59.410734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lst_test = !ls /kaggle/input/test-rsna-dataset/10008\n# for i in lst_test:\n#     img = cv2.imread(\"/kaggle/input/test-rsna-dataset/10008/\"+i, 0)\n#     img = pre_img(img)\n#     transform = Compose([Resize(height=227,width=227,always_apply=True),\n#                                       ToTensorV2()])\n#     img = transform(image=img)['image']\n#     print(model(img.unsqueeze(0)).argmax().item())","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-02-06T13:56:59.413158Z","iopub.execute_input":"2023-02-06T13:56:59.413574Z","iopub.status.idle":"2023-02-06T13:56:59.422343Z","shell.execute_reply.started":"2023-02-06T13:56:59.413536Z","shell.execute_reply":"2023-02-06T13:56:59.421449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### f. Save file predict to submit\n","metadata":{}},{"cell_type":"code","source":"# df = pd.read_csv(csv_test_RSNA)\n# lst_test = ['/kaggle/input/rsna-breast-cancer-detection/test_images/' + str(df.patient_id[i]) + '/' + str(df.image_id[i]) + '.dcm' for i in range(len(df))]\n","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-02-06T13:56:59.424093Z","iopub.execute_input":"2023-02-06T13:56:59.42445Z","iopub.status.idle":"2023-02-06T13:56:59.433161Z","shell.execute_reply.started":"2023-02-06T13:56:59.424416Z","shell.execute_reply":"2023-02-06T13:56:59.432229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checkpoint=torch.load(PATH)\n# print(checkpoint['state_dict'])","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-02-06T13:56:59.434816Z","iopub.execute_input":"2023-02-06T13:56:59.435179Z","iopub.status.idle":"2023-02-06T13:56:59.444146Z","shell.execute_reply.started":"2023-02-06T13:56:59.435146Z","shell.execute_reply":"2023-02-06T13:56:59.443199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pre_img(img):\n    img_clahe = equalize_adapthist(img, clip_limit=0.04, nbins=256)\n    return img_clahe\ndef ROI(img):\n    # smooth image\n    blur = cv2.GaussianBlur(img,(5, 5), 0)\n    # threshold image\n    threshold = ((blur > np.mean(img))*255).astype(np.uint8)\n    # Apply the Component analysis function\n    analysis = cv2.connectedComponentsWithStats(threshold, 4, cv2.CV_32S)\n    (totalLabels, label_ids, values, centroid) = analysis\n    # Loop through each component\n    crop_img = None\n    for i in range(1, totalLabels):\n        area = values[i, cv2.CC_STAT_AREA]  \n        s = img.size*0.1\n        if (area > s):\n            # Create a new image for bounding boxes\n            new_img = img.copy()\n            # Now extract the coordinate points\n            x = values[i, cv2.CC_STAT_LEFT]\n            y = values[i, cv2.CC_STAT_TOP]\n            w = values[i, cv2.CC_STAT_WIDTH]\n            h = values[i, cv2.CC_STAT_HEIGHT]\n            crop_img = new_img[y:y+h, x:x+w]\n    return crop_img\n    \ndef make_predict_file(csv_path, model):\n    df = pd.read_csv(csv_path)\n    df['prediction_id'] = [str(df.patient_id[i]) + '_' + str(df.laterality[i]) for i in range(len(df))]\n    if(csv_path.split('/')[-1] =='test.csv'):\n        lst_test = ['/kaggle/input/rsna-breast-cancer-detection/test_images/' + str(df.patient_id[i]) + '/' + str(df.image_id[i]) + '.dcm' for i in range(len(df))]\n    else:\n        lst_test = ['/kaggle/input/rsna-breast-cancer-detection/train_images/' + str(df.patient_id[i]) + '/' + str(df.image_id[i]) + '.dcm' for i in range(len(df))]\n    lst_predict = []\n    for path in tqdm(lst_test):\n        # read describe image\n#         dcm = pydicom.dcmread(path)\n        try:\n            dcm = dicomsdl.open(path)\n            img = dcm.pixelData(storedvalue=False) # storedvalue = True for int16 return otherwise float32\n    #         img = cv2.imread(path, 0)\n            # read image\n    #         img = dcm.pixel_array\n            img = ((img-img.min())/(img.max()-img.min()))*255\n            # invert pixel\n            if dcm.PhotometricInterpretation=='MONOCHROME1':\n                img = img.max()-img\n\n            img = ROI(img)\n            img = img.astype('uint8')\n            img = cv2.resize(img, (227, 227))\n            img = pre_img(img)\n    #             plt.imshow(img)\n            transform = Compose([Resize(height=227,width=227,always_apply=True),\n                                              ToTensorV2()])\n            img = transform(image=img)['image']\n            lst_predict.append(model(img.unsqueeze(0)).argmax().item())\n        except:\n            pass\n    submission = pd.DataFrame({'prediction_id': df.prediction_id, 'cancer':lst_predict}).groupby(\"prediction_id\").mean().reset_index()\n    return submission","metadata":{"execution":{"iopub.status.busy":"2023-02-06T13:56:59.445651Z","iopub.execute_input":"2023-02-06T13:56:59.446035Z","iopub.status.idle":"2023-02-06T13:56:59.511592Z","shell.execute_reply.started":"2023-02-06T13:56:59.445988Z","shell.execute_reply":"2023-02-06T13:56:59.510584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = '/kaggle/input/rsna-breast-cancer-detection/train.csv'\nmake_predict_file(csv_test_RSNA, model).to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-02-06T13:56:59.513158Z","iopub.execute_input":"2023-02-06T13:56:59.513822Z","iopub.status.idle":"2023-02-06T13:57:02.950255Z","shell.execute_reply.started":"2023-02-06T13:56:59.513783Z","shell.execute_reply":"2023-02-06T13:57:02.949277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# make_predict_file(path, model)","metadata":{"execution":{"iopub.status.busy":"2023-02-06T13:57:02.951908Z","iopub.execute_input":"2023-02-06T13:57:02.952637Z","iopub.status.idle":"2023-02-06T13:57:02.957793Z","shell.execute_reply.started":"2023-02-06T13:57:02.952582Z","shell.execute_reply":"2023-02-06T13:57:02.956424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}