{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":22307,"databundleVersionId":1502524,"sourceType":"competition"},{"sourceId":1488548,"sourceType":"datasetVersion","datasetId":873734}],"dockerImageVersionId":30120,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -U git+git://github.com/lilohuang/PyTurboJPEG.git\n!pip install torch-optimizer\n!pip install PyTurboJPEG\n!conda install -c conda-forge gdcm -y\n!pip install efficientnet-pytorch","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-10-01T15:37:58.524346Z","iopub.execute_input":"2021-10-01T15:37:58.52475Z","iopub.status.idle":"2021-10-01T15:39:56.928791Z","shell.execute_reply.started":"2021-10-01T15:37:58.524666Z","shell.execute_reply":"2021-10-01T15:39:56.927612Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"!pip install -U git+git://github.com/lilohuang/PyTurboJPEG.git\n!pip install torch-optimizer\n!pip install PyTurboJPEG\n!conda install -c conda-forge gdcm -y\n!pip install efficientnet-pytorch","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np # linear algebra\nfrom sklearn.metrics import classification_report, confusion_matrix\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport glob\nimport cv2\nimport os\nfrom matplotlib import pyplot as plt\nfrom torch.utils.data import TensorDataset, DataLoader,Dataset,SubsetRandomSampler\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nfrom skimage.color import gray2rgb\nimport functools\nimport torch\nfrom torch import Tensor\nfrom torchvision import transforms\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.autograd import Variable\nfrom torch.nn.parameter import Parameter\nfrom torch.optim import lr_scheduler\nimport torch_optimizer as optim\nfrom torch.utils.data import DataLoader, Dataset\nfrom torch.utils.data.sampler import SequentialSampler\nfrom tqdm.auto import tqdm\n\n#from matplotlib import animation, rc\nfrom plotly.subplots import make_subplots\nimport plotly.graph_objs as go\nimport pydicom as dcm\n#import gdcm\nimport time\nfrom turbojpeg import TurboJPEG, TJPF_GRAY, TJSAMP_GRAY","metadata":{"execution":{"iopub.status.busy":"2021-10-01T15:39:56.933055Z","iopub.execute_input":"2021-10-01T15:39:56.933356Z","iopub.status.idle":"2021-10-01T15:40:00.941112Z","shell.execute_reply.started":"2021-10-01T15:39:56.933323Z","shell.execute_reply":"2021-10-01T15:40:00.940034Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from efficientnet_pytorch import EfficientNet","metadata":{"execution":{"iopub.status.busy":"2021-10-01T15:40:00.943879Z","iopub.execute_input":"2021-10-01T15:40:00.944363Z","iopub.status.idle":"2021-10-01T15:40:00.954017Z","shell.execute_reply.started":"2021-10-01T15:40:00.944306Z","shell.execute_reply":"2021-10-01T15:40:00.952822Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"batch_size = 128\nno_epochs = 2\ntrain_path = '../input/rsna-str-pulmonary-embolism-detection/train.csv'\njpeg_path = '../input/rsna-str-pe-detection-jpeg-256/train-jpegs'\ntrain_df=pd.read_csv(train_path)\ndf_main=train_df.copy()\nprint(df_main.head())\nprint(len(glob.glob(jpeg_path+'/*')))\ndf_main.SOPInstanceUID.nunique()\nprint('shape',df_main.shape)\nlist_of_jpgs= [i for i in range(1,10)]\ncolumns_only_for_info = [\"qa_motion\",\"qa_contrast\",\"flow_artifact\",\"true_filling_defect_not_pe\"]\n\ntrain_df.iloc[list_of_jpgs,:]\n\ntarget_columns = ['StudyInstanceUID','SeriesInstanceUID','SOPInstanceUID','pe_present_on_image', 'negative_exam_for_pe', 'rv_lv_ratio_gte_1', \n                  'rv_lv_ratio_lt_1','leftsided_pe', 'chronic_pe','rightsided_pe', \n                  'acute_and_chronic_pe', 'central_pe', 'indeterminate']\ntrain_df[target_columns]","metadata":{"execution":{"iopub.status.busy":"2021-10-01T15:40:00.956261Z","iopub.execute_input":"2021-10-01T15:40:00.95681Z","iopub.status.idle":"2021-10-01T15:40:06.965387Z","shell.execute_reply.started":"2021-10-01T15:40:00.956766Z","shell.execute_reply":"2021-10-01T15:40:06.964304Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class jpg_dataset(Dataset):\n    def __init__(self, root_dir, df,  transforms = None):\n        \n        super().__init__()\n        self.df = df[target_columns]\n\n        self.root_dir = root_dir\n        self.transforms = transforms\n\n    def __len__(self):\n        return self.df.shape[0]\n\n    def __getitem__(self, ndx):\n        row = self.df.iloc[ndx,:]\n        in_file = open(glob.glob(f\"{root_dir}/{row[0]}/{row[1]}/*{row[2]}.jpg\")[0], 'rb')\n        img = jpeg.decode(in_file.read())\n        in_file.close()\n        label = row[3:].astype(int)\n        label[2:] = label[2:] if label[0]==1 else 0\n        #img /= 255.0\n        \n        #Retriving class label\n\n        \n        #Applying transforms on image\n        if self.transforms:\n            img = self.transforms(image=img)['image']\n        \n\n        return (img,label.values)","metadata":{"execution":{"iopub.status.busy":"2021-10-01T15:40:06.967579Z","iopub.execute_input":"2021-10-01T15:40:06.9679Z","iopub.status.idle":"2021-10-01T15:40:06.979658Z","shell.execute_reply.started":"2021-10-01T15:40:06.96787Z","shell.execute_reply":"2021-10-01T15:40:06.978323Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"root_dir=jpeg_path\njpeg = TurboJPEG()\n\n\nrow = train_df.iloc[0,:]\nrow\n\njpg_data = jpg_dataset(root_dir=jpeg_path,df=train_df)\n\n\ndataloader = DataLoader(jpg_data, batch_size=2,\n                        shuffle=True, num_workers=0)\n#dataloader\n\nfor i_batch, sample_batched in enumerate(dataloader):\n    image=sample_batched[0]\n    print(i_batch, image.size())\n\n    # observe 4th batch and stop.\n    if i_batch == 3:\n        plt.figure()\n        plt.imshow(image[0])\n        plt.axis('off')\n        plt.ioff()\n        plt.show()\n        break","metadata":{"execution":{"iopub.status.busy":"2021-10-01T15:40:06.981208Z","iopub.execute_input":"2021-10-01T15:40:06.981846Z","iopub.status.idle":"2021-10-01T15:40:08.079359Z","shell.execute_reply.started":"2021-10-01T15:40:06.981802Z","shell.execute_reply":"2021-10-01T15:40:08.078093Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(jpg_data)\nprint(dataloader)","metadata":{"execution":{"iopub.status.busy":"2021-10-01T15:40:08.081012Z","iopub.execute_input":"2021-10-01T15:40:08.081449Z","iopub.status.idle":"2021-10-01T15:40:08.088775Z","shell.execute_reply.started":"2021-10-01T15:40:08.081403Z","shell.execute_reply":"2021-10-01T15:40:08.087442Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"StudyInstanceUID = list(set(train_df['StudyInstanceUID']))\nprint(len(StudyInstanceUID))\nt_df = train_df[train_df['StudyInstanceUID'].isin(StudyInstanceUID[0:800])]\nv_df = train_df[train_df['StudyInstanceUID'].isin(StudyInstanceUID[800:1000])]\nprint('t_df',len(t_df))\nprint('v_df',len(v_df))\n\nt_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-10-01T15:40:08.092671Z","iopub.execute_input":"2021-10-01T15:40:08.093666Z","iopub.status.idle":"2021-10-01T15:40:08.521139Z","shell.execute_reply.started":"2021-10-01T15:40:08.093619Z","shell.execute_reply":"2021-10-01T15:40:08.520113Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_transforms = {\n    'train': A.Compose(\n    [\n        #A.SmallestMaxSize(max_size=160),\n        #A.ShiftScaleRotate(shift_limit=0.05, scale_limit=0.05, rotate_limit=15, p=0.5),\n        A.RandomCrop(height=64, width=64),\n        #A.RGBShift(r_shift_limit=15, g_shift_limit=15, b_shift_limit=15, p=0.5),\n        A.RandomBrightnessContrast(p=0.5),\n        A.Normalize(mean=(0.485, 0.456, 0.406), std=(0.229, 0.224, 0.225)),\n        ToTensorV2(),\n    ]\n),\n    'val': A.Compose(\n    [\n        A.SmallestMaxSize(max_size=160),\n        A.CenterCrop(height=64, width=64),\n        A.Normalize(mean=(0.485, 0.456, 0.406), std=(0.229, 0.224, 0.225)),\n        ToTensorV2(),\n    ]\n),\n}\n","metadata":{"execution":{"iopub.status.busy":"2021-10-01T15:40:08.523115Z","iopub.execute_input":"2021-10-01T15:40:08.523513Z","iopub.status.idle":"2021-10-01T15:40:08.532727Z","shell.execute_reply.started":"2021-10-01T15:40:08.523469Z","shell.execute_reply":"2021-10-01T15:40:08.531074Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Initializing Datasets and Dataloaders...\")\n\n# Create training and validation datasets\nimage_datasets = {}\nimage_datasets['train'] = jpg_dataset(root_dir=jpeg_path,df=t_df, transforms = data_transforms['train'])\nimage_datasets['val'] = jpg_dataset(root_dir=jpeg_path,df=v_df, transforms = data_transforms['val'])\n\n# Create training and validation dataloaders\ndataloaders_dict = {x: torch.utils.data.DataLoader(image_datasets[x], batch_size=batch_size, shuffle=True, num_workers=16, pin_memory=True) for x in ['train', 'val']}\n\ndataset_sizes = {x: len(image_datasets[x]) for x in ['train', 'val']}\n\ndataset_sizes['train']\nimage_datasets['train'][0]\n","metadata":{"execution":{"iopub.status.busy":"2021-10-01T15:40:08.534693Z","iopub.execute_input":"2021-10-01T15:40:08.535187Z","iopub.status.idle":"2021-10-01T15:40:08.646319Z","shell.execute_reply.started":"2021-10-01T15:40:08.535144Z","shell.execute_reply":"2021-10-01T15:40:08.645257Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from efficientnet_pytorch import EfficientNet","metadata":{"execution":{"iopub.status.busy":"2021-10-01T15:40:08.649805Z","iopub.execute_input":"2021-10-01T15:40:08.650149Z","iopub.status.idle":"2021-10-01T15:40:08.657939Z","shell.execute_reply.started":"2021-10-01T15:40:08.650118Z","shell.execute_reply":"2021-10-01T15:40:08.656727Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class EfficientNetEncoderHead(nn.Module):\n    \n    def __init__(self, depth, num_classes):\n        super(EfficientNetEncoderHead, self).__init__()\n        self.depth = depth\n        self.base = EfficientNet.from_pretrained(f'efficientnet-b{self.depth}')\n        self.avg_pool = nn.AdaptiveAvgPool2d(1)\n        self.output_filter = self.base._fc.in_features\n        self.classifier = nn.Linear(self.output_filter, num_classes)\n    def forward(self, x):\n        x = self.base.extract_features(x)\n        x = self.avg_pool(x).squeeze(-1).squeeze(-1)\n        x = self.classifier(x)\n        return x\n\nmodel = EfficientNetEncoderHead(depth=0, num_classes=10)\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\nmodel = model.to(device)","metadata":{"execution":{"iopub.status.busy":"2021-10-01T15:40:08.659921Z","iopub.execute_input":"2021-10-01T15:40:08.660859Z","iopub.status.idle":"2021-10-01T15:40:15.037994Z","shell.execute_reply.started":"2021-10-01T15:40:08.660811Z","shell.execute_reply":"2021-10-01T15:40:15.036919Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_model(model, criterion, optimizer, scheduler, no_epochs=25):\n    \n    since = time.time()\n\n    best_model_wts = copy.deepcopy(model.state_dict())\n    best_acc = 0.0\n\n    for epoch in range(no_epochs):\n        print('Epoch {}/{}'.format(epoch, no_epochs - 1))\n        print('-' * 10)\n        \n        \n        # Each epoch has a training and validation phase\n        for phase in ['train', 'val']:\n            if phase == 'train':\n                model.train()  # Set model to training mode\n            else:\n                model.eval()   # Set model to evaluate mode\n\n            running_loss = 0.0\n            current_loss_mean = 0.0\n            running_corrects = 0\n            \n\n            # Iterate over data.\n            tqdm_loader = tqdm(dataloaders_dict[phase])\n       \n            for batch_idx, (inputs,labels) in enumerate(tqdm_loader):\n                #inputs = inputs.to(device)\n                #labels = labels.to(device)\n                \n                inputs, labels = inputs.cuda().float(), labels.cuda().float() \n\n                # zero the parameter gradients\n                optimizer.zero_grad()\n                \n                # forward\n                # track history only if in train\n                with torch.set_grad_enabled(phase == 'train'):\n\n                    outputs = model(inputs)\n                    \n                    loss = criterion(outputs.float(), labels)\n                    \n\n                    # backward + optimize only if in training phase\n                    if phase == 'train':\n                        \n                        loss.backward()\n                        optimizer.step()\n                    \n                    # statistics\n                \n                running_loss += loss.item() * inputs.size(0)\n                current_loss_mean = (current_loss_mean * batch_idx + loss) / (batch_idx + 1)\n                tqdm_loader.set_description('loss: {:.4}'.format(\n                    current_loss_mean))\n                running_corrects += torch.sum(outputs == labels.data)\n            if phase == 'train':\n                scheduler.step()\n\n            epoch_loss = running_loss / dataset_sizes[phase]\n            score = 1-current_loss_mean\n            print('metric {}'.format(score))\n            \n            epoch_acc = running_corrects.double() / dataset_sizes[phase]\n\n            \n            \n            \n            correct = 0\n            total = 0\n            # since we're not training, we don't need to calculate the gradients for our outputs\n            # the class with the highest energy is what we choose as prediction\n           \n            _, predicted = torch.max(outputs,1)\n            total += labels.size(0)\n            print(\"total\",total)\n           \n            for j in range(len(predicted)):\n                #print(correct)\n                if((labels[predicted[j]] == labels[j]).all()):\n                    correct +=1 \n                    #print(correct)\n                \n            print(correct)\n            print('{} Loss: {:.4f} Acc: {:.4f}'.format(\n                phase, epoch_loss, 100 * correct / total))\n            print('Accuracy of the network on the  test images: %d %%' % (\n                100 * correct / total))\n            _, predicted = torch.max(outputs, 1)\n            print('Predicted: ', ' '.join('%5s' % labels[predicted[j]]\n                                for j in range(4)))\n            #end add predicction\n            \n            # deep copy the model\n            if phase == 'val' and epoch_acc > best_acc:\n                best_acc = epoch_acc\n                best_model_wts = copy.deepcopy(model.state_dict())\n\n\n    time_elapsed = time.time() - since\n    print('Training complete in {:.0f}m {:.0f}s'.format(\n        time_elapsed // 60, time_elapsed % 60))\n    print('Best val Acc: {:4f}'.format(best_acc))\n\n    # load best model weights\n    model.load_state_dict(best_model_wts)\n    torch.save(model.state_dict(),f'model_new_all.bin')\n    return model\n\ndef radam(parameters, lr=1e-3, betas=(0.9, 0.999), eps=1e-8, weight_decay=0):\n    if isinstance(betas, str):\n        betas = eval(betas)\n    return optim.RAdam(parameters,\n                      lr=lr,\n                      betas=betas,\n                      eps=eps,\n                      weight_decay=weight_decay)\n\ncriterion = torch.nn.BCEWithLogitsLoss()\n\n# Observe that all parameters are being optimized\n\noptimizer = radam(model.parameters(), lr=1e-3, betas=(0.9,0.999), eps=1e-3, weight_decay=1e-4)\nscheduler = lr_scheduler.CosineAnnealingLR(optimizer, T_max=len(dataloaders_dict['train'])*no_epochs, eta_min=1e-6) # dataloader should be train one - change later\n\n# Decay LR by a factor of 0.1 every 7 epochs\nexp_lr_scheduler = lr_scheduler.StepLR(optimizer, step_size=7, gamma=0.1)\nimport copy\ntrain_model(model, criterion, optimizer, scheduler, no_epochs=no_epochs)\n#visualize_model(model_ft)","metadata":{"execution":{"iopub.status.busy":"2021-10-01T15:40:15.03977Z","iopub.execute_input":"2021-10-01T15:40:15.040222Z","iopub.status.idle":"2021-10-01T16:09:36.235627Z","shell.execute_reply.started":"2021-10-01T15:40:15.040175Z","shell.execute_reply":"2021-10-01T16:09:36.232932Z"},"trusted":true},"outputs":[],"execution_count":null}]}