{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"! pip install timm","metadata":{"execution":{"iopub.status.busy":"2022-12-22T04:22:47.359753Z","iopub.execute_input":"2022-12-22T04:22:47.360503Z","iopub.status.idle":"2022-12-22T04:22:57.034291Z","shell.execute_reply.started":"2022-12-22T04:22:47.360405Z","shell.execute_reply":"2022-12-22T04:22:57.032932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport cv2\nimport glob\nimport matplotlib.pyplot as plt\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nfrom sklearn import metrics\nfrom sklearn.model_selection import StratifiedGroupKFold\nimport timm\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import DataLoader, Dataset, SubsetRandomSampler, random_split\nfrom torchvision.transforms import ToTensor\nfrom torchvision import transforms\nfrom tqdm.notebook import tqdm\n\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-22T04:22:57.042204Z","iopub.execute_input":"2022-12-22T04:22:57.042564Z","iopub.status.idle":"2022-12-22T04:22:59.220841Z","shell.execute_reply.started":"2022-12-22T04:22:57.042501Z","shell.execute_reply":"2022-12-22T04:22:59.219507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torch.__version__","metadata":{"execution":{"iopub.status.busy":"2022-12-22T04:22:59.237318Z","iopub.execute_input":"2022-12-22T04:22:59.237998Z","iopub.status.idle":"2022-12-22T04:22:59.248405Z","shell.execute_reply.started":"2022-12-22T04:22:59.237958Z","shell.execute_reply":"2022-12-22T04:22:59.247345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load train csv and check out the data\ntrain_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntrain_df['path'] = (\n    train_df['patient_id'].astype(str)\n    + '_'\n    + train_df['image_id'].astype(str)\n    + '.png'\n)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-22T04:22:59.250311Z","iopub.execute_input":"2022-12-22T04:22:59.250654Z","iopub.status.idle":"2022-12-22T04:22:59.427605Z","shell.execute_reply.started":"2022-12-22T04:22:59.250625Z","shell.execute_reply":"2022-12-22T04:22:59.426575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ax = sns.countplot(train_df['cancer'])\nax.set_title('Label Distribution')\nax.grid()","metadata":{"execution":{"iopub.status.busy":"2022-12-22T04:22:59.428892Z","iopub.execute_input":"2022-12-22T04:22:59.429624Z","iopub.status.idle":"2022-12-22T04:22:59.633936Z","shell.execute_reply.started":"2022-12-22T04:22:59.429567Z","shell.execute_reply":"2022-12-22T04:22:59.633211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load image\nimage = cv2.imread('/kaggle/input/rsna-breast-cancer-1024-pngs/output/10006_1459541791.png')\nplt.imshow(image)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T04:23:39.162672Z","iopub.execute_input":"2022-12-22T04:23:39.163059Z","iopub.status.idle":"2022-12-22T04:23:39.54275Z","shell.execute_reply.started":"2022-12-22T04:23:39.163026Z","shell.execute_reply":"2022-12-22T04:23:39.541792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image","metadata":{"execution":{"iopub.status.busy":"2022-12-22T04:23:39.544604Z","iopub.execute_input":"2022-12-22T04:23:39.545451Z","iopub.status.idle":"2022-12-22T04:23:39.554868Z","shell.execute_reply.started":"2022-12-22T04:23:39.545412Z","shell.execute_reply":"2022-12-22T04:23:39.553779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create torch dataset & dataset loader for RSNA Breast Cancer detection\nclass RSNABreastDataset(Dataset):\n    def __init__(self, df, image_dir, transforms=None):\n        # Set the image path\n        self.image_paths = df['path'].values\n        self.image_dir = image_dir\n        \n        # Set whether we should apply augmentations / transformations\n        self.transforms = transforms\n        \n        # Set up the training labels\n        self.labels = df['cancer'].values\n        \n    def __len__(self):\n        # Essentially tells us how many images are in the\n        # dataset\n        return len(self.image_paths)\n    \n    def __getitem__(self, idx):\n        # Get the images - can read in multiple images\n        # at ones with cv2\n        image = cv2.imread(\n            os.path.join(self.image_dir, self.image_paths[idx])\n        )\n        \n        if self.transforms:\n            image = self.transforms(image) #['image']\n            \n        labels = torch.tensor([self.labels[idx]], dtype=torch.float)\n        \n        # This is not specified in the code but this is probably for\n        # weighting the class imbalance\n        w = torch.tensor([1])\n        \n        return image, labels, w\n    \ntrain_transform = transforms.Compose([\n    transforms.ToTensor(),\n])","metadata":{"execution":{"iopub.status.busy":"2022-12-22T04:23:39.637757Z","iopub.execute_input":"2022-12-22T04:23:39.638132Z","iopub.status.idle":"2022-12-22T04:23:39.646778Z","shell.execute_reply.started":"2022-12-22T04:23:39.638101Z","shell.execute_reply":"2022-12-22T04:23:39.645436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Setup the train data loader\n# image_train_path = '/kaggle/input/rsna-breast-cancer-1024-pngs/output'\nimage_train_path = '/kaggle/input/rsna-breast-cancer-512-pngs/'\ntrain_data = RSNABreastDataset(\n    df=train_df,\n    image_dir=image_train_path,\n    transforms=train_transform,\n)\n\ntrain_data_loader = DataLoader(train_data, batch_size=4, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T04:23:40.869342Z","iopub.execute_input":"2022-12-22T04:23:40.869738Z","iopub.status.idle":"2022-12-22T04:23:40.875797Z","shell.execute_reply.started":"2022-12-22T04:23:40.869704Z","shell.execute_reply":"2022-12-22T04:23:40.874769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Iterate through the data and let's visualize!\ntrain_features, train_labels, train_weights = next(iter(train_data_loader))","metadata":{"execution":{"iopub.status.busy":"2022-12-22T04:23:41.288489Z","iopub.execute_input":"2022-12-22T04:23:41.289585Z","iopub.status.idle":"2022-12-22T04:23:41.339052Z","shell.execute_reply.started":"2022-12-22T04:23:41.289514Z","shell.execute_reply":"2022-12-22T04:23:41.337849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_features.shape","metadata":{"execution":{"iopub.status.busy":"2022-12-22T04:23:42.143894Z","iopub.execute_input":"2022-12-22T04:23:42.14584Z","iopub.status.idle":"2022-12-22T04:23:42.152483Z","shell.execute_reply.started":"2022-12-22T04:23:42.145793Z","shell.execute_reply":"2022-12-22T04:23:42.151329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 2, figsize=(10, 10))\naxes = axes.flatten()\n\nfor index, ax in enumerate(axes):\n    image = train_features[index, :, :, :].squeeze().transpose(0, 2)\n    label = int(train_labels[index])\n    ax.imshow(image)\n    ax.set_title(f'Has Cancer = {label}')\n    \nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-12-22T04:23:42.642012Z","iopub.execute_input":"2022-12-22T04:23:42.642392Z","iopub.status.idle":"2022-12-22T04:23:43.640598Z","shell.execute_reply.started":"2022-12-22T04:23:42.642359Z","shell.execute_reply":"2022-12-22T04:23:43.639483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_devices():\n    \"\"\"\n    Function to get GPU devices if available. If there are\n    no GPU devices available then use CPU\n    \"\"\"\n    # Default device\n    device = torch.device(\"cpu\")\n\n    # Get available GPUs\n    n_gpus = torch.cuda.device_count()\n    if n_gpus > 0:\n        print(n_gpus)\n        gpu_name_list = [f\"cuda:{device}\" for device in range(n_gpus)]\n\n        # NOTE: For now we will only use the first GPU\n        # but we might want to consider distributed GPUs in the future\n        device = torch.device(gpu_name_list[0])\n\n    return device\n\ndevice = get_devices()\ndevice","metadata":{"execution":{"iopub.status.busy":"2022-12-22T04:23:43.818784Z","iopub.execute_input":"2022-12-22T04:23:43.819159Z","iopub.status.idle":"2022-12-22T04:23:43.869347Z","shell.execute_reply.started":"2022-12-22T04:23:43.819129Z","shell.execute_reply":"2022-12-22T04:23:43.868104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device.type","metadata":{"execution":{"iopub.status.busy":"2022-12-22T04:23:44.973372Z","iopub.execute_input":"2022-12-22T04:23:44.9741Z","iopub.status.idle":"2022-12-22T04:23:44.980686Z","shell.execute_reply.started":"2022-12-22T04:23:44.974062Z","shell.execute_reply":"2022-12-22T04:23:44.979641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class RSNAEfficientnetModel(nn.Module):\n    def __init__(self, pretrained=True):\n        self.pretrained = pretrained\n        super().__init__()\n        \n        # Setup the model\n        self.model = timm.create_model(\n            'efficientnet_b0',\n            pretrained=True,\n            drop_rate=0.3,\n            drop_path_rate=0.2,\n            global_pool='avg',\n            num_classes=1\n        )\n        # Set up the first convolutional layer to have stride 1\n        self.model.conv_stem.stride = (1, 1)\n        \n    def forward(self, x):\n        x = self.model(x)\n        return x\n    \n# Train one epoch  \ndef train_one_epoch(model, data_loader, epoch, criterion, optimizer, scaler, device):\n    \"\"\"\n    Function to train one epoch of our image classifier\n    \"\"\"\n    # Put model in training model\n    model.train()\n\n    # Training loss\n    all_predictions = []\n    all_labels = []\n    train_loss = 0\n\n    for index, (images, labels, weights) in tqdm(enumerate(data_loader)):\n        # Make sure the images are compatible with the model\n        # and located on the right device\n        images = images.to(device=device)\n        labels = labels.to(device=device)\n\n        # Zero out the gradients\n        optimizer.zero_grad()\n\n        # Used mixed precision for modeling\n        with torch.autocast(device_type=device.type, dtype=torch.float16):\n            # Get the predictions\n            predictions = model(images)\n            sigmoid_predictions = torch.sigmoid(predictions)\n\n            # Calculate the loss\n            loss = criterion(predictions, labels)\n\n        # Take the backward step\n        scaler.scale(loss).backward()\n\n        # Step through the optimizer\n        scaler.step(optimizer)\n        scaler.update()\n\n        # accumulate the loss\n        train_loss += loss.item()\n        \n        # Accumlate labels as well\n        all_predictions.append(sigmoid_predictions.detach().cpu().numpy())\n        all_labels.append(labels.detach().cpu().numpy())\n\n    print(f\"Epoch = {epoch} / loss = {train_loss / len(data_loader)}\")\n    all_predictions = np.concatenate(all_predictions).flatten().astype(np.float64)\n    all_labels = np.concatenate(all_labels).flatten().astype(np.float64)\n    return all_predictions, all_labels\n    \n    \ndef evaluate_model(model, data_loader, metric, device):\n    \"\"\"\n    Function to evaluate the model. The metrics can be a dictionary\n    that will run multiple metrics for the model\n    \"\"\"\n    model.eval()\n\n    # Make sure we turn off the ability to change / update gradients\n    with torch.no_grad():\n        predictions, labels = build_predictions(model, data_loader, device)\n\n        # Assume sklearn metrics\n        metric_value = metric(labels, predictions)\n        print(f\"Evaluation metric = {metric_value}\")\n        \n        \ndef build_predictions(model, data_loader, device):\n    \"\"\"\n    Function to simply build predictions for the new model\n    \"\"\"\n    all_predictions = []\n    all_labels = []\n    model.eval()\n\n    # Make sure we turn off the ability to change / update gradients\n    # TODO: will have to do something if the labels are empty\n    # I think this will be present in the data loader\n    with torch.no_grad():\n        for index, (images, labels, weights) in tqdm(enumerate(data_loader)):\n            images = images.to(device=device)\n\n            # Get the predictions\n            predictions = model(images)\n\n            # With classification do not forget that we need\n            # to add a sigmoid layer afterwards for probablistic\n            # predictions\n            predictions = torch.sigmoid(predictions)\n\n            all_predictions.append(predictions.detach().cpu().numpy())\n            all_labels.append(labels.detach().cpu().numpy())\n\n    return np.concatenate(all_predictions), np.concatenate(all_labels)\n\n# Probablistic F1 score\ndef pfbeta(labels, predictions, beta):\n    \"\"\"\n    Function to calculate the probablistic F1 score\n    \"\"\"\n    y_true_count = 0\n    ctp = 0\n    cfp = 0\n\n    for idx in range(len(labels)):\n        prediction = min(max(predictions[idx], 0), 1)\n        if labels[idx]:\n            y_true_count += 1\n            ctp += prediction\n        else:\n            cfp += prediction\n\n    beta_squared = beta * beta\n    c_precision = ctp / (ctp + cfp)\n    c_recall = ctp / y_true_count\n    if c_precision > 0 and c_recall > 0:\n        result = (\n            (1 + beta_squared)\n            * (c_precision * c_recall)\n            / (beta_squared * c_precision + c_recall)\n        )\n        return result\n    else:\n        return 0\n\n# Setup the model and then save the weights as they are currently / no training\nmodel = RSNAEfficientnetModel(pretrained=True).to(device=device)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T04:23:56.001146Z","iopub.execute_input":"2022-12-22T04:23:56.001545Z","iopub.status.idle":"2022-12-22T04:23:57.98395Z","shell.execute_reply.started":"2022-12-22T04:23:56.001495Z","shell.execute_reply":"2022-12-22T04:23:57.982909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# When we do cross validation we will do it as group k-fold because in the test data we will have\n# completly different patients\nn_splits = 5\ngroup_kfold = StratifiedGroupKFold(n_splits=5)\n\nindexes = np.arange(len(train_df))\ny = train_df['cancer'].values\ngroups = train_df['patient_id'].values\n\n# Set up optimizer\nscaler = torch.cuda.amp.GradScaler()\noptimizer = torch.optim.AdamW(model.parameters(), lr=1e-3)\n\nfor fold, (train_idx, val_idx) in enumerate(group_kfold.split(X=indexes, y=y, groups=groups)):\n    print(f'Fold = {fold + 1}')\n    \n    # Set up the train and validation data\n    train_sampler = SubsetRandomSampler(train_idx)\n    val_sampler = SubsetRandomSampler(val_idx)\n    \n    X_train_loader = DataLoader(train_data, batch_size=12, sampler=train_sampler)\n    X_val_loader = DataLoader(train_data, batch_size=8, sampler=val_sampler)\n    \n    print(f'Number of samples in X_train_loader = {len(X_train_loader)}')\n    print(f'Number of samples in X_val_loader = {len(X_val_loader)}')\n    \n    # Set up loss function\n    train_labels = X_train_loader.dataset.labels\n    pos_samples = train_labels.sum()\n    pos_weight = torch.tensor(int(len(train_labels) / pos_samples))\n    criterion = nn.BCEWithLogitsLoss(pos_weight=pos_weight).to(device=device)\n    \n    # Set model into training mode\n    model.train()\n    \n    for epoch in range(5):\n        # Train a single epoch\n        train_predictions, train_labels = train_one_epoch(\n            model=model,\n            data_loader=X_train_loader,\n            epoch=epoch,\n            criterion=criterion,\n            optimizer=optimizer,\n            scaler=scaler,\n            device=device,\n        )\n        \n        # Print training loss for epoch\n        training_log_loss = metrics.log_loss(train_labels, train_predictions)\n        training_pfbeta = pfbeta(train_labels, train_predictions, beta=1.0)\n        training_aucroc = metrics.roc_auc_score(train_labels, train_predictions)\n        print(f'Training log loss = {training_log_loss}')\n        print(f'Training pfbeta = {training_pfbeta}')\n        print(f'Training aucroc = {training_aucroc}')\n        \n        # Evaluate the model\n        val_predictions, val_labels = build_predictions(\n            model=model,\n            data_loader=X_val_loader,\n            device=device\n        )\n        \n        # Print the evaluation metric\n        val_log_loss = metrics.log_loss(val_labels, val_predictions)\n        val_pfbeta = pfbeta(val_labels, val_predictions, beta=1.0)\n        val_aucroc = metrics.roc_auc_score(val_labels, val_predictions)\n        print(f'Val log loss = {val_log_loss}')\n        print(f'Val pfbeta = {val_pfbeta}')\n        print(f'Val aucroc = {val_aucroc}')\n    \n    if fold == 0:\n        break","metadata":{"execution":{"iopub.status.busy":"2022-12-22T04:28:21.936788Z","iopub.execute_input":"2022-12-22T04:28:21.937166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Next Steps 🛰\n-  Finish the evaluation function for evaluating our model ✅\n\n-  Learn how to save and load weights for torch models - https://pytorch.org/tutorials/beginner/saving_loading_models.html ✅\n\n-  Look into and understand the evaluation metric - probablistic F1 score - this is an extension of the traditional F score that accepts probabilities instead of binary classifications\n\n-  Get the submission up and running - this will be one of the more difficult parts because it uses things I have never seen before (nvidia-dali)","metadata":{}},{"cell_type":"code","source":"metrics.log_loss(train_labels.flatten().astype(np.float64), train_predictions.flatten().astype(np.float64))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = '/kaggle/working/model_weights'\npath_exists_bool = os.path.exists(path)\n\n# Set up path if it does not exists\nif not path_exists_bool:\n    os.makedirs(path)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the model\ntorch_model_weights_filename = 'model_weights.pth'\ntorch.save(model.state_dict(), os.path.join(path, torch_model_weights_filename))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}